From 0cddf96c489b67341879526f54ce738a9ba05c8f Mon Sep 17 00:00:00 2001 From: Leitwolf11 Date: Wed, 30 Sep 2026 15:59:02 +0300 Subject: [PATCH 1/6] Separate source lowering and differential fuzz campaigns; share verified frontend input across optimizer modes; require measured execution and RSS evidence in campaign reports; strengthen deterministic oracle smoke tests and document sanitizer boundaries --- Compiler/Fuzzing/BUILD.bazel | 7 +++ .../Corpus/differential/arithmetic.seed | 1 + Compiler/Fuzzing/SourceFuzz.cpp | 25 ++++---- Compiler/Fuzzing/SourceFuzzSmoke.cpp | 10 +++ Compiler/Fuzzing/SourceLibFuzzer.cpp | 3 + Documents/FUZZING.md | 25 ++++++-- Documents/TESTING.md | 19 +++--- REUSE.toml | 6 ++ helpers/internal/development/cli_test.go | 4 +- helpers/internal/development/fuzz.go | 25 +++++--- helpers/internal/development/fuzz_report.go | 61 +++++++++++++++++++ .../internal/development/fuzz_report_test.go | 46 ++++++++++++++ 12 files changed, 195 insertions(+), 37 deletions(-) create mode 100644 Compiler/Fuzzing/Corpus/differential/arithmetic.seed create mode 100644 helpers/internal/development/fuzz_report.go create mode 100644 helpers/internal/development/fuzz_report_test.go diff --git a/Compiler/Fuzzing/BUILD.bazel b/Compiler/Fuzzing/BUILD.bazel index 0fbbce17..0d366e82 100644 --- a/Compiler/Fuzzing/BUILD.bazel +++ b/Compiler/Fuzzing/BUILD.bazel @@ -70,3 +70,10 @@ cc_binary( defines = ["VXS_SOURCE_FUZZ_STAGE=2"], deps = [":source_fuzz_harness"], ) + +cc_binary( + name = "differential_fuzzer", + srcs = ["SourceLibFuzzer.cpp"], + defines = ["VXS_SOURCE_FUZZ_STAGE=3"], + deps = [":source_fuzz_harness"], +) diff --git a/Compiler/Fuzzing/Corpus/differential/arithmetic.seed b/Compiler/Fuzzing/Corpus/differential/arithmetic.seed new file mode 100644 index 00000000..55a38a55 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/differential/arithmetic.seed @@ -0,0 +1 @@ +arithmetic-oracle-seed diff --git a/Compiler/Fuzzing/SourceFuzz.cpp b/Compiler/Fuzzing/SourceFuzz.cpp index 55b7aab2..559411bc 100644 --- a/Compiler/Fuzzing/SourceFuzz.cpp +++ b/Compiler/Fuzzing/SourceFuzz.cpp @@ -174,23 +174,15 @@ namespace Visual::XSharp::Fuzzing } [[nodiscard]] auto - CompileVariant(std::span source, + CompileVariant(std::span coreBytes, bool optimizeXpp, bool optimizeXmm) -> Llvm::Artifact { - const auto compiled = CompileSource(source); - if (!compiled.succeeded() - || compiled.kind != Frontend::OutputKind::CoreWire) - llvm::report_fatal_error( - llvm::Twine("generated arithmetic source was rejected by " - "the frontend")); - Visual::XSharp::Pipeline::Options options; options.optimize_xpp = optimizeXpp; options.optimize_xmm = optimizeXmm; const auto pipeline - = Visual::XSharp::Pipeline::ConsumeCore(compiled.bytes, - options); + = Visual::XSharp::Pipeline::ConsumeCore(coreBytes, options); if (!pipeline || !pipeline.llvm) llvm::report_fatal_error(llvm::Twine( "differential source failed a verified compiler pipeline: " @@ -288,8 +280,17 @@ namespace Visual::XSharp::Fuzzing const auto bytes = std::span( reinterpret_cast(source.data()), source.size()); - const auto unoptimized = CompileVariant(bytes, false, false); - const auto optimized = CompileVariant(bytes, true, true); + const auto compiled = CompileSource(bytes); + if (!compiled.succeeded() + || compiled.kind != Frontend::OutputKind::CoreWire) + llvm::report_fatal_error( + llvm::Twine("generated arithmetic source was rejected by " + "the frontend")); + // The comparison varies native optimizers, so both paths start from + // the same frontend result. Recompiling identical source adds no + // independent evidence and repeats work in the expensive oracle. + const auto unoptimized = CompileVariant(compiled.bytes, false, false); + const auto optimized = CompileVariant(compiled.bytes, true, true); constexpr std::string_view kReferenceModule = "vxs-fuzz-reference"; constexpr std::string_view kOptimizedModule = "vxs-fuzz-optimized"; const auto referenceValue = Invoke(unoptimized, kReferenceModule); diff --git a/Compiler/Fuzzing/SourceFuzzSmoke.cpp b/Compiler/Fuzzing/SourceFuzzSmoke.cpp index 72090b04..a28e48b6 100644 --- a/Compiler/Fuzzing/SourceFuzzSmoke.cpp +++ b/Compiler/Fuzzing/SourceFuzzSmoke.cpp @@ -24,5 +24,15 @@ main() Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(emptySource); Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(source); Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(expressionSeed); + Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(emptySource); + // Constant seeds force leaves, full-depth addition, subtraction and + // multiplication. Exercise the independent oracle before a mutation + // campaign so a missing generated-code route cannot appear as success. + constexpr std::array selectors{ 252U, 253U, 254U, 255U }; + for (const auto selector : selectors) + { + const std::array seed{ selector }; + Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(seed); + } return 0; } diff --git a/Compiler/Fuzzing/SourceLibFuzzer.cpp b/Compiler/Fuzzing/SourceLibFuzzer.cpp index bc326475..3f933312 100644 --- a/Compiler/Fuzzing/SourceLibFuzzer.cpp +++ b/Compiler/Fuzzing/SourceLibFuzzer.cpp @@ -20,7 +20,10 @@ LLVMFuzzerTestOneInput(const std::uint8_t *data, std::size_t size) #elif VXS_SOURCE_FUZZ_STAGE == 1 Visual::XSharp::Fuzzing::ExerciseParser(input); #elif VXS_SOURCE_FUZZ_STAGE == 2 + // Arbitrary source and generated arithmetic have independent corpora and + // time budgets. An invalid source mutation should not pay for two JITs. Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(input); +#elif VXS_SOURCE_FUZZ_STAGE == 3 Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(input); #else # error Unsupported VXS_SOURCE_FUZZ_STAGE diff --git a/Documents/FUZZING.md b/Documents/FUZZING.md index a9c22acf..3a07dec6 100644 --- a/Documents/FUZZING.md +++ b/Documents/FUZZING.md @@ -14,15 +14,23 @@ memory safety or complete language coverage. | `wire_fuzzer` | Core, private CorePrep transport, Xpp and Xmm; bounded decoding, semantic verification and equal encode/decode round trips | First-party C++ codecs and verifiers | | `lexer_fuzzer` | Arbitrary bytes through the frontend lexer ABI; complete token/diagnostic evaluation | Native ABI bridge, not GHC-generated lexer branches | | `parser_fuzzer` | Arbitrary bytes through syntax analysis; complete AST/diagnostic evaluation | Native ABI bridge, not GHC-generated parser branches | -| `source_llvm_fuzzer` | Source through Core/CorePrep, Xpp/Xmm verification and LLVM lowering; generated arithmetic differential oracle | First-party C++ pipeline and JIT bridge | +| `source_llvm_fuzzer` | Arbitrary source through Core/CorePrep, Xpp/Xmm verification and LLVM lowering | First-party C++ pipeline | +| `differential_fuzzer` | Generated arithmetic compiled with native optimizers disabled/enabled and compared with an independent evaluator | First-party C++ pipeline and JIT bridge | -The source oracle independently evaluates bounded generated arithmetic, compiles -it with Xpp/Xmm optimizations both disabled and enabled, executes both verified +The differential oracle independently evaluates bounded generated arithmetic, +compiles its source once, lowers the same verified Core with Xpp/Xmm +optimizations both disabled and enabled, executes both verified artifacts through ORC, and compares all three results. This detects miscompiles within that generated subset; it is not an oracle for arbitrary Visual X# programs. Invalid source is a normal rejection, whereas internal failures and verified-model inconsistencies fail the campaign. +Arbitrary source and generated arithmetic use separate corpora and equal +per-target time budgets. This lets source mutations reach native lowering +without repeatedly creating two ORC sessions for unrelated generated code. +The differential generator consumes at most 31 selector bytes; its 64-byte +input limit keeps mutations near the bytes that influence the program. + GHC frontend code and prebuilt LLVM dependencies do not receive Clang native coverage instrumentation. Lexer/parser execution must not be presented as coverage-guided exploration of their Haskell implementation. Haskell tests and @@ -54,6 +62,7 @@ Each campaign has a 30-second per-input timeout and a finite input length: | Lexer | 65536 | 1024 | | Parser | 65536 | 1536 | | Source/LLVM | 65536 | 4096 | +| Differential arithmetic | 64 | 4096 | ASan intentionally retains freed allocations in quarantine. Fuzz-only settings bound this cache to 64 MiB, with a 256 KiB thread-local cache; the nonzero @@ -64,7 +73,7 @@ normal quarantine settings. ## Corpus synchronization and reports Wire seeds come from production writers, so format-version changes do not leave -handwritten supposedly valid documents behind. Lexer, parser and source seeds +handwritten supposedly valid documents behind. Lexer, parser, source and differential seeds come from `Compiler/Fuzzing/Corpus/`. Set `VXS_FUZZ_CORPUS` to retain mutation corpora across local runs. Updated versioned seeds are added without overwriting older discovered inputs; conflicting contents under a stable hash fail closed. @@ -72,7 +81,11 @@ older discovered inputs; conflicting contents under a stable hash fail closed. GitHub Actions restores a per-platform corpus cache and saves a unique cache version for each run. It also uploads campaign artifacts on success or failure. Reports contain the target, duration, RSS limit, selected sanitizer, native -coverage ownership and result. Logs include execution counts and peak RSS. +coverage ownership and result. Structured reports also include executed inputs, +average executions per second, new corpus entries, slowest input time and peak +RSS from libFuzzer's final counters. A successful process without a complete +final report or with zero executed inputs fails the campaign gate. Failed +processes retain their original logs even if final counters are unavailable. Failed work directories are preserved for diagnosis; successful CI reports are retained for artifact upload. @@ -113,5 +126,5 @@ protection must require these actual checks. CLI argument generation, project/lockfile inputs, persistent REPL sessions and broader generated language programs need independent oracles. Deep semantic cases, ownership concurrency and frontend feedback-guided coverage are not -established by the four existing targets. Expand these deliberately instead of +established by the five existing targets. Expand these deliberately instead of equating a green workflow with completion of the entire security program. diff --git a/Documents/TESTING.md b/Documents/TESTING.md index 2b566d1b..b92f2f80 100644 --- a/Documents/TESTING.md +++ b/Documents/TESTING.md @@ -118,7 +118,8 @@ go run ./helpers/cmd/develop doctor go run ./helpers/cmd/develop test ``` -The command executes 18 Catch3 binaries and one C11 ABI contract executable on +The command executes 18 Catch3 binaries, one C11 ABI contract executable and +one source/differential fuzz smoke executable on Windows 10/11, macOS Sequoia/Tahoe, Ubuntu 26.04 LTS, and Fedora 43. Bazel selects the host configuration automatically; no public test instruction requires `--config`. @@ -142,14 +143,18 @@ labels shorten iteration, but they do not replace the full native gate. See Run native memory diagnostics through the same entry point: ```powershell -go run ./helpers/cmd/develop sanitize address +go run ./helpers/cmd/develop sanitize address-undefined ``` -macOS and Linux additionally support `sanitize undefined` and `sanitize thread`. Each sanitizer instruments both compilation and -linking and runs the complete native suite set rather than merely proving that instrumented objects compile. +Windows, macOS and Linux support combined ASan/UBSan and their individual +`sanitize address` / `sanitize undefined` profiles. macOS and native Linux also +support the separate `sanitize thread` profile. Each profile verifies real +intentional failures before running the complete native suite set. See +[Fuzzing](FUZZING.md) for the Windows and Fedora TSan boundaries. -The Compiler Tier 1/2/3 workflows also run a separate deterministic wire-mutation smoke target. It is not included in the 16 -component-owned `develop.go test` suites. Run it directly when a Core, CorePrep, Xpp, or Xmm decoder changes: +The Compiler Tier 1/2/3 workflows also run a separate deterministic wire-mutation +smoke target. It is separate from the component-owned `develop test` suite set. +Run it directly when a Core, CorePrep, Xpp, or Xmm decoder changes: ```powershell bazelisk build //Compiler/Fuzzing:wire_fuzz_smoke @@ -336,7 +341,7 @@ or below 1500 lines. A simple review aid is: ```powershell $extensions = '*.hs','*.cpp','*.hpp','*.hh','*.kt','*.kts','*.go','*.java' -Get-ChildItem Compiler,Interactive,ProjectSystem,Analyzer,Formatter,Linter,scripts -Recurse -File -Include $extensions | +Get-ChildItem Compiler,Interactive,ProjectSystem,Analyzer,Formatter,Linter,helpers -Recurse -File -Include $extensions | Where-Object { (Get-Content -LiteralPath $_.FullName).Count -gt 1500 } | Select-Object FullName ``` diff --git a/REUSE.toml b/REUSE.toml index 05f4ce9e..79fcb978 100644 --- a/REUSE.toml +++ b/REUSE.toml @@ -8,3 +8,9 @@ path = [".bazelversion", "MODULE.bazel.lock", ".github/brand/visual-xsharp-socia precedence = "aggregate" SPDX-FileCopyrightText = "2026 Progmasoft " SPDX-License-Identifier = "MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1" + +[[annotations]] +path = ["Compiler/Fuzzing/Corpus/**"] +precedence = "aggregate" +SPDX-FileCopyrightText = "2026 Progmasoft " +SPDX-License-Identifier = "MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1" diff --git a/helpers/internal/development/cli_test.go b/helpers/internal/development/cli_test.go index 6eb14b6f..7e9be798 100644 --- a/helpers/internal/development/cli_test.go +++ b/helpers/internal/development/cli_test.go @@ -64,8 +64,8 @@ func TestFuzzBuildPlansSeparateSmokeAndRuntimeMain(t *testing.T) { drivers++ } } - if drivers != 4 { - t.Fatalf("expected four campaign drivers: %v", campaign) + if drivers != 5 { + t.Fatalf("expected five campaign drivers: %v", campaign) } if sanitizer != "" { for _, plan := range [][]string{smoke, campaign} { diff --git a/helpers/internal/development/fuzz.go b/helpers/internal/development/fuzz.go index 2c33cc10..0ede95a6 100644 --- a/helpers/internal/development/fuzz.go +++ b/helpers/internal/development/fuzz.go @@ -39,7 +39,7 @@ func macOSFuzzerRuntime(root string) (string, error) { return matches[0], nil } -// Smoke programs own main; only the four campaign drivers may link libFuzzer's +// Smoke programs own main; only campaign drivers may link libFuzzer's // main. Sharing one global fuzz profile with both groups duplicates main on // Linux, where -fsanitize=fuzzer pulls the driver in unconditionally. func fuzzBuildArguments(configuration, sanitizerConfiguration, macRuntime string) ([]string, []string) { @@ -57,7 +57,8 @@ func fuzzBuildArguments(configuration, sanitizerConfiguration, macRuntime string "//Compiler/Fuzzing:wire_fuzzer", "//Compiler/Fuzzing:lexer_fuzzer", "//Compiler/Fuzzing:parser_fuzzer", - "//Compiler/Fuzzing:source_llvm_fuzzer") + "//Compiler/Fuzzing:source_llvm_fuzzer", + "//Compiler/Fuzzing:differential_fuzzer") return smoke, campaign } @@ -108,7 +109,7 @@ func runFuzzCampaign(repository string, currentHost host, runner commandRunner, if err := os.Mkdir(artifacts, 0o700); err != nil { return fmt.Errorf("could not create fuzz artifact directory %q: %w", work, err) } - stageNames := []string{"wire", "lexer", "parser", "source"} + stageNames := []string{"wire", "lexer", "parser", "source", "differential"} for _, stage := range stageNames { corpus := filepath.Join(corpusRoot, stage) if err := os.MkdirAll(corpus, 0o700); err != nil { @@ -119,11 +120,7 @@ func runFuzzCampaign(repository string, currentHost host, runner commandRunner, if stage == "wire" { continue } - seedName := stage - if stage == "source" { - seedName = "source" - } - if err := syncSeedCorpus(filepath.Join(repository, "Compiler", "Fuzzing", "Corpus", seedName), corpus); err != nil { + if err := syncSeedCorpus(filepath.Join(repository, "Compiler", "Fuzzing", "Corpus", stage), corpus); err != nil { return fmt.Errorf("could not synchronize the versioned %s seed corpus: %w", stage, err) } } @@ -187,11 +184,15 @@ func runFuzzCampaign(repository string, currentHost host, runner commandRunner, {"lexer_fuzzer", "lexer", "65536", "1024"}, {"parser_fuzzer", "parser", "65536", "1536"}, {"source_llvm_fuzzer", "source", "65536", "4096"}, + // The depth-four generator consumes at most 31 selector bytes. Bound + // mutations close to this semantic input rather than evolving unused + // tails alongside the expensive two-module execution oracle. + {"differential_fuzzer", "differential", "64", "4096"}, } var records []map[string]any for _, target := range targets { fuzzer := filepath.Join(repository, "bazel-bin", "Compiler", "Fuzzing", target.binary+currentHost.executable) - if strings.Contains(target.binary, "lexer") || strings.Contains(target.binary, "parser") || strings.Contains(target.binary, "source_llvm") { + if target.binary != "wire_fuzzer" { if err := copyFile(frontendLibrary, filepath.Join(filepath.Dir(fuzzer), filepath.Base(frontendLibrary)), 0o755); err != nil { return fmt.Errorf("could not stage frontend for %s; preserved %q: %w", target.binary, work, err) } @@ -216,7 +217,11 @@ func runFuzzCampaign(repository string, currentHost host, runner commandRunner, return fmt.Errorf("could not preserve campaign log: %w", err) } fmt.Print(output) - records = append(records, map[string]any{"target": target.binary, "seconds": duration, "rss_limit_mb": target.rssLimit, "sanitizer": sanitizerConfiguration, "native_coverage": true, "haskell_native_coverage": false, "success": runErr == nil}) + statistics, statisticsErr := parseFuzzStatistics(output) + if runErr == nil && statisticsErr != nil { + runErr = statisticsErr + } + records = append(records, map[string]any{"target": target.binary, "seconds": duration, "rss_limit_mb": target.rssLimit, "sanitizer": sanitizerConfiguration, "native_coverage": true, "haskell_native_coverage": false, "statistics": statistics, "success": runErr == nil}) report, err := json.MarshalIndent(records, "", " ") if err != nil { return err diff --git a/helpers/internal/development/fuzz_report.go b/helpers/internal/development/fuzz_report.go new file mode 100644 index 00000000..ff1eba50 --- /dev/null +++ b/helpers/internal/development/fuzz_report.go @@ -0,0 +1,61 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "fmt" + "strconv" + "strings" +) + +// Keep libFuzzer's counters beside the requested budget. Exit zero alone does +// not establish that a campaign executed any inputs or produced a final report. +type fuzzStatistics struct { + ExecutedUnits uint64 `json:"executed_units"` + AverageExecPerSec uint64 `json:"average_exec_per_second"` + NewUnitsAdded uint64 `json:"new_units_added"` + SlowestUnitSeconds uint64 `json:"slowest_unit_seconds"` + PeakRSSMiB uint64 `json:"peak_rss_mib"` +} + +func parseFuzzStatistics(output string) (*fuzzStatistics, error) { + statistics := &fuzzStatistics{} + fields := map[string]*uint64{ + "number_of_executed_units": &statistics.ExecutedUnits, + "average_exec_per_sec": &statistics.AverageExecPerSec, + "new_units_added": &statistics.NewUnitsAdded, + "slowest_unit_time_sec": &statistics.SlowestUnitSeconds, + "peak_rss_mb": &statistics.PeakRSSMiB, + } + seen := make(map[string]bool, len(fields)) + for _, line := range strings.Split(output, "\n") { + statistic, ok := strings.CutPrefix(strings.TrimSpace(line), "stat::") + if !ok { + continue + } + name, value, ok := strings.Cut(statistic, ":") + destination, recognized := fields[name] + if !recognized { + continue // Future libFuzzer counters do not change this contract. + } + if !ok || seen[name] { + return nil, fmt.Errorf("invalid or duplicate libFuzzer statistic %q", name) + } + parsed, err := strconv.ParseUint(strings.TrimSpace(value), 10, 64) + if err != nil { + return nil, fmt.Errorf("invalid libFuzzer statistic %q: %w", name, err) + } + *destination = parsed + seen[name] = true + } + for name := range fields { + if !seen[name] { + return nil, fmt.Errorf("libFuzzer did not report %q", name) + } + } + if statistics.ExecutedUnits == 0 || statistics.PeakRSSMiB == 0 { + return nil, fmt.Errorf("libFuzzer reported no executed input or no measured RSS") + } + return statistics, nil +} diff --git a/helpers/internal/development/fuzz_report_test.go b/helpers/internal/development/fuzz_report_test.go new file mode 100644 index 00000000..f87aa3ad --- /dev/null +++ b/helpers/internal/development/fuzz_report_test.go @@ -0,0 +1,46 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "strings" + "testing" +) + +const fuzzFinalStatistics = "INFO: seed corpus loaded\r\n" + + "stat::number_of_executed_units: 155\r\n" + + "stat::average_exec_per_sec: 5\r\n" + + "stat::new_units_added: 139\r\n" + + "stat::slowest_unit_time_sec: 0\r\n" + + "stat::peak_rss_mb: 196\r\n" + +func TestFuzzStatisticsPreserveExecutionEvidence(t *testing.T) { + statistics, err := parseFuzzStatistics(fuzzFinalStatistics + "stat::future_counter: 42\n") + if err != nil { + t.Fatal(err) + } + if statistics.ExecutedUnits != 155 || statistics.AverageExecPerSec != 5 || + statistics.NewUnitsAdded != 139 || statistics.SlowestUnitSeconds != 0 || statistics.PeakRSSMiB != 196 { + t.Fatalf("lost campaign evidence: %+v", statistics) + } +} + +func TestFuzzStatisticsRejectIncompleteOrContradictoryReports(t *testing.T) { + for name, report := range map[string]string{ + "missing": "INFO: clean exit without final statistics\n", + "truncated": strings.ReplaceAll(fuzzFinalStatistics, "stat::peak_rss_mb: 196\r\n", ""), + "duplicate": fuzzFinalStatistics + "stat::number_of_executed_units: 1\n", + "zero inputs": strings.ReplaceAll(fuzzFinalStatistics, "units: 155", "units: 0"), + "negative": strings.ReplaceAll(fuzzFinalStatistics, "units: 155", "units: -1"), + "overflow": strings.ReplaceAll(fuzzFinalStatistics, "units: 155", "units: 18446744073709551616"), + "fraction": strings.ReplaceAll(fuzzFinalStatistics, "units: 155", "units: 1.5"), + "no RSS": strings.ReplaceAll(fuzzFinalStatistics, "196", "0"), + } { + t.Run(name, func(t *testing.T) { + if _, err := parseFuzzStatistics(report); err == nil { + t.Fatal("accepted a report without reliable execution evidence") + } + }) + } +} From f875bac8b10b5a61710a52ecfce1ba4f522cab2d Mon Sep 17 00:00:00 2001 From: Leitwolf11 Date: Thu, 1 Oct 2026 22:21:57 +0300 Subject: [PATCH 2/6] Fix native Core-to-CorePrep control-flow miscompiles and complete the fuzz campaign inventory The native Core-to-CorePrep adapter built loop and logical-operator control flow that diverged from the Haskell CorePrep lowering: - a for-loop update region fell through to its own entry instead of the condition, so every for loop that reached its update never terminated; - a while condition was folded into the incoming block, so each back-edge re-executed the statements preceding the loop, including its initializers; - the canonicalizing comparison of a numeric for condition was dropped, which made valid programs fail CorePrep verification; - && and || were lowered as eager two-operand instructions, so a guarded division trapped and a guarded recursion overflowed the stack. The adapter now threads one block cursor through expressions and statements. Loop headers are dedicated blocks, the fallthrough successor of a loop region is separate from its continue target, and short-circuit operators become a branch with an initialized Boolean result slot, matching the Haskell lowering. Regression coverage asserts exact CorePrep edges for for, while, do/while, nested loops and short-circuit operators, and executes each form through CorePrep, Xpp, Xmm and LLVM with both native optimizer settings against values computed by host code. One existing test that pinned the eager LogicalAnd shape now checks the per-operand canonical comparisons in the lazy shape. The differential oracle gains while and guarded-recursion shapes, and the source smoke sweeps every shape across trip counts 0 through 11. Fuzzing: add component-owned CLI, project registry, REPL and AARC ownership libFuzzer targets with versioned seed corpora, a bounded project registry decoder entry point with its tests, and an HPC-guided Haskell frontend engine with feedback and mutation policy tests. The developer helper shares one target inventory, terminates the whole process tree when a smoke, libFuzzer or HPC process exceeds its watchdog, isolates the HPC tick file per stage, reads the HPC report from the engine's report file, and adds fuzz-thread, a separate ThreadSanitizer campaign for threaded targets that fails on hosts without a ThreadSanitizer runtime. CI runs the ThreadSanitizer ownership campaign on macOS and native Ubuntu and the fuzz feedback tests with the Haskell layers. The fuzzing, Core IR, testing and developer recipe documents describe the verified inventory, limits and remaining coverage boundaries. Co-Authored-By: Claude Opus 5.5 --- .github/workflows/fuzzing.yml | 10 + .github/workflows/language-layers.yml | 5 + .gitignore | 3 + Compiler/Backend/LLVM/Tests/BUILD.bazel | 2 + .../Backend/LLVM/Tests/LoopExecutionTests.cpp | 400 +++++++ .../LLVM/Tests/ShortCircuitExecutionTests.cpp | 318 ++++++ Compiler/Cli/Arguments/Options.hpp | 9 + Compiler/Cli/Fuzzing/BUILD.bazel | 7 + Compiler/Cli/Fuzzing/CliFuzzer.cpp | 54 + Compiler/Core/CorePrep/Prepare.cpp | 1013 ++++++++--------- Compiler/Core/Tests/BUILD.bazel | 2 + Compiler/Core/Tests/CorePipelineTests.cpp | 36 +- Compiler/Core/Tests/LoopLoweringTests.cpp | 453 ++++++++ .../Core/Tests/ShortCircuitLoweringTests.cpp | 327 ++++++ Compiler/Fuzzing/Corpus/cli/options.seed | 3 + Compiler/Fuzzing/Corpus/ownership/race.seed | 1 + Compiler/Fuzzing/Corpus/project/registry.seed | 1 + Compiler/Fuzzing/Corpus/repl/history.seed | 1 + Compiler/Fuzzing/SourceFuzz.cpp | 94 +- Compiler/Fuzzing/SourceFuzzSmoke.cpp | 22 + Compiler/Haskell/Driver/Fuzzing/Feedback.hs | 39 + Compiler/Haskell/Driver/Fuzzing/Main.hs | 182 +++ Compiler/Haskell/Driver/Fuzzing/Mutation.hs | 76 ++ Compiler/Haskell/Driver/Fuzzing/Tests.hs | 39 + .../Driver/visual-xsharp-compiler.cabal | 28 + .../ProjectSystem/Bridge/Fuzzing/BUILD.bazel | 7 + .../Bridge/Fuzzing/ProjectFuzzer.cpp | 81 ++ .../ProjectSystem/Bridge/ProjectDriver.cpp | 20 +- .../ProjectSystem/Bridge/ProjectDriver.hpp | 11 + .../ProjectSystem/Bridge/Tests/BUILD.bazel | 7 + .../Bridge/Tests/RegistryTests.cpp | 91 ++ Compiler/Runtime/AARC/Fuzzing/BUILD.bazel | 7 + .../Runtime/AARC/Fuzzing/OwnershipFuzzer.cpp | 96 ++ Documents/CORE-IR.md | 30 + Documents/DEVELOPER-RECIPES.md | 1 + Documents/FUZZING.md | 111 +- Documents/TESTING.md | 2 +- Interactive/Fuzzing/BUILD.bazel | 7 + Interactive/Fuzzing/ReplFuzzer.cpp | 70 ++ helpers/internal/development/build.go | 1 + helpers/internal/development/clean.go | 1 + helpers/internal/development/cli.go | 3 +- helpers/internal/development/cli_test.go | 6 +- helpers/internal/development/commands.go | 8 + helpers/internal/development/fuzz.go | 147 +-- helpers/internal/development/fuzz_haskell.go | 138 +++ .../development/fuzz_haskell_report_test.go | 47 + .../internal/development/fuzz_haskell_test.go | 70 ++ helpers/internal/development/fuzz_targets.go | 43 + helpers/internal/development/fuzz_thread.go | 125 ++ .../internal/development/fuzz_thread_test.go | 44 + helpers/internal/development/process.go | 71 +- .../development/process_alive_unix_test.go | 13 + .../development/process_alive_windows_test.go | 19 + helpers/internal/development/process_test.go | 90 ++ helpers/internal/development/process_unix.go | 32 + .../internal/development/process_windows.go | 36 + justfile | 4 + 58 files changed, 3890 insertions(+), 674 deletions(-) create mode 100644 Compiler/Backend/LLVM/Tests/LoopExecutionTests.cpp create mode 100644 Compiler/Backend/LLVM/Tests/ShortCircuitExecutionTests.cpp create mode 100644 Compiler/Cli/Fuzzing/BUILD.bazel create mode 100644 Compiler/Cli/Fuzzing/CliFuzzer.cpp create mode 100644 Compiler/Core/Tests/LoopLoweringTests.cpp create mode 100644 Compiler/Core/Tests/ShortCircuitLoweringTests.cpp create mode 100644 Compiler/Fuzzing/Corpus/cli/options.seed create mode 100644 Compiler/Fuzzing/Corpus/ownership/race.seed create mode 100644 Compiler/Fuzzing/Corpus/project/registry.seed create mode 100644 Compiler/Fuzzing/Corpus/repl/history.seed create mode 100644 Compiler/Haskell/Driver/Fuzzing/Feedback.hs create mode 100644 Compiler/Haskell/Driver/Fuzzing/Main.hs create mode 100644 Compiler/Haskell/Driver/Fuzzing/Mutation.hs create mode 100644 Compiler/Haskell/Driver/Fuzzing/Tests.hs create mode 100644 Compiler/ProjectSystem/Bridge/Fuzzing/BUILD.bazel create mode 100644 Compiler/ProjectSystem/Bridge/Fuzzing/ProjectFuzzer.cpp create mode 100644 Compiler/ProjectSystem/Bridge/Tests/BUILD.bazel create mode 100644 Compiler/ProjectSystem/Bridge/Tests/RegistryTests.cpp create mode 100644 Compiler/Runtime/AARC/Fuzzing/BUILD.bazel create mode 100644 Compiler/Runtime/AARC/Fuzzing/OwnershipFuzzer.cpp create mode 100644 Interactive/Fuzzing/BUILD.bazel create mode 100644 Interactive/Fuzzing/ReplFuzzer.cpp create mode 100644 helpers/internal/development/fuzz_haskell.go create mode 100644 helpers/internal/development/fuzz_haskell_report_test.go create mode 100644 helpers/internal/development/fuzz_haskell_test.go create mode 100644 helpers/internal/development/fuzz_targets.go create mode 100644 helpers/internal/development/fuzz_thread.go create mode 100644 helpers/internal/development/fuzz_thread_test.go create mode 100644 helpers/internal/development/process_alive_unix_test.go create mode 100644 helpers/internal/development/process_alive_windows_test.go create mode 100644 helpers/internal/development/process_test.go create mode 100644 helpers/internal/development/process_unix.go create mode 100644 helpers/internal/development/process_windows.go diff --git a/.github/workflows/fuzzing.yml b/.github/workflows/fuzzing.yml index 501f742b..779a65f5 100644 --- a/.github/workflows/fuzzing.yml +++ b/.github/workflows/fuzzing.yml @@ -152,6 +152,11 @@ jobs: env: VXS_FUZZ_CORPUS: ${{ runner.temp }}/vxs-fuzz-corpus run: go run ./helpers/cmd/develop fuzz + - name: Run bounded ThreadSanitizer ownership campaign + shell: bash + env: + VXS_FUZZ_CORPUS: ${{ runner.temp }}/vxs-fuzz-corpus + run: go run ./helpers/cmd/develop fuzz-thread - name: Run nightly ASan and UBSan stress campaign if: github.event_name == 'schedule' shell: bash @@ -216,6 +221,11 @@ jobs: env: VXS_FUZZ_CORPUS: ${{ runner.temp }}/vxs-fuzz-corpus run: go run ./helpers/cmd/develop fuzz + - name: Run bounded ThreadSanitizer ownership campaign + shell: bash + env: + VXS_FUZZ_CORPUS: ${{ runner.temp }}/vxs-fuzz-corpus + run: go run ./helpers/cmd/develop fuzz-thread - name: Run nightly ASan and UBSan stress campaign if: github.event_name == 'schedule' shell: bash diff --git a/.github/workflows/language-layers.yml b/.github/workflows/language-layers.yml index 9fb2874c..a3781254 100644 --- a/.github/workflows/language-layers.yml +++ b/.github/workflows/language-layers.yml @@ -54,6 +54,11 @@ jobs: working-directory: Compiler run: cabal test visual-xsharp-compiler-tests + - name: Test frontend fuzz feedback and mutation policy + shell: pwsh + working-directory: Compiler + run: cabal test fuzz-feedback-tests + - name: Check Haskell package metadata shell: pwsh run: | diff --git a/.gitignore b/.gitignore index 498ca605..2c1efa36 100644 --- a/.gitignore +++ b/.gitignore @@ -6,6 +6,7 @@ bin/ xide_api/ xide/ .codex/ +.claude/ .agents/ .gradle/ .kotlin/ @@ -13,6 +14,8 @@ xide/ node_modules/ dist/ dist-newstyle/ +dist-fuzz-coverage/ +*.tix coverage/ *.profraw *.profdata diff --git a/Compiler/Backend/LLVM/Tests/BUILD.bazel b/Compiler/Backend/LLVM/Tests/BUILD.bazel index 7bd32f56..e389d5ec 100644 --- a/Compiler/Backend/LLVM/Tests/BUILD.bazel +++ b/Compiler/Backend/LLVM/Tests/BUILD.bazel @@ -8,6 +8,8 @@ cc_binary( "CallableInvocationTests.cpp", "LLVMBackendTests.cpp", "JitSessionTests.cpp", + "LoopExecutionTests.cpp", + "ShortCircuitExecutionTests.cpp", ], deps = [ "//Compiler/Backend/LLVM:llvm_backend", diff --git a/Compiler/Backend/LLVM/Tests/LoopExecutionTests.cpp b/Compiler/Backend/LLVM/Tests/LoopExecutionTests.cpp new file mode 100644 index 00000000..e38cabaa --- /dev/null +++ b/Compiler/Backend/LLVM/Tests/LoopExecutionTests.cpp @@ -0,0 +1,400 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Backend/LLVM.hpp" +#include "Visual/XSharp/Core/IR.hpp" +#include "Visual/XSharp/Core/Wire.hpp" +#include "Visual/XSharp/Pipeline.hpp" + +// Executable loop regressions. Every case is lowered from structured Core +// through CorePrep, Xpp, Xmm and LLVM, run in ORC, and compared with a value +// computed by an ordinary host loop written in this file. Comparing the +// optimized and unoptimized pipelines with each other is not sufficient: a +// defect in the shared Core-to-CorePrep adapter makes both agree on the same +// wrong answer, or on the same non-terminating program. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Llvm = Visual::XSharp::Backend::LLVM; + namespace Pipeline = Visual::XSharp::Pipeline; + namespace Prepared = visual_xsharp::core; + + constexpr std::uint64_t kTotal = 2U; + constexpr std::uint64_t kIndex = 3U; + constexpr std::uint64_t kInner = 4U; + constexpr std::int64_t kLimits = 13; + + [[nodiscard]] auto + Spelling(std::uint64_t id) -> std::u32string + { + return id == kTotal ? U"total" : id == kIndex ? U"index" : U"inner"; + } + + [[nodiscard]] auto + Integer(std::int64_t value) -> Core::Expression + { + return Core::Expression::Constant(value, Core::Type::int64()); + } + + [[nodiscard]] auto + Variable(std::uint64_t id) -> Core::Expression + { + return Core::Expression::Variable({ id, Spelling(id) }, + Core::Type::int64()); + } + + [[nodiscard]] auto + Compare(Core::Primitive operation, std::uint64_t id, std::int64_t value) + -> Core::Expression + { + return Core::Expression::InvokePrimitive( + operation, + { Variable(id), Integer(value) }, + Core::Type::boolean()); + } + + [[nodiscard]] auto + Add(std::uint64_t destination, Core::Expression value) -> Core::Statement + { + return Core::Statement::Assign( + { destination, Spelling(destination) }, + Core::Expression::InvokePrimitive( + Core::Primitive::Add, + { Variable(destination), std::move(value) }, + Core::Type::int64())); + } + + [[nodiscard]] auto + Increment(std::uint64_t id) -> Core::Statement + { + return Add(id, Integer(1)); + } + + [[nodiscard]] auto + When(std::uint64_t id, std::int64_t value, Core::Statement then) + -> Core::Statement + { + return Core::Statement::If(Compare(Core::Primitive::Equal, id, value), + { std::move(then) }, + {}); + } + + [[nodiscard]] auto + Declare(std::uint64_t id) -> Core::Statement + { + return Core::Statement::Bind( + { { id, Spelling(id) }, Core::Type::int64(), true, Integer(0) }); + } + + /// `int total = 0; int index = 0; ; return total;` + [[nodiscard]] auto + LoopModule(Core::Statement loop) -> Core::Module + { + Core::Function function{ + { 1U, U"Evaluate" }, + {}, + Core::Type::int64(), + { Declare(kTotal), + Declare(kIndex), + std::move(loop), + Core::Statement::Return(Variable(kTotal)) }, + }; + return { { U"Loops" }, { std::move(function) } }; + } + + /// True when every CorePrep block can still reach a function return. + /// A loop region that cannot leave is rejected before it is executed, so + /// a control-flow regression fails this suite instead of hanging it. + [[nodiscard]] auto + EveryBlockReachesReturn(const Prepared::Function &function) -> bool + { + const auto successors = [](const Prepared::Block &block) { + std::vector targets; + if (block.terminator.kind == Prepared::Terminator::Kind::Jump) + targets = { block.terminator.true_target }; + if (block.terminator.kind == Prepared::Terminator::Kind::Branch) + targets = { block.terminator.true_target, + block.terminator.false_target }; + return targets; + }; + // Backward fixed point from the return blocks. + std::vector returning; + for (const auto &block : function.blocks) + if (block.terminator.kind == Prepared::Terminator::Kind::Return) + returning.push_back(block.id); + for (bool changed = true; changed;) + { + changed = false; + for (const auto &block : function.blocks) + { + if (std::ranges::find(returning, block.id) != returning.end()) + continue; + const auto targets = successors(block); + if (std::ranges::any_of(targets, [&](const auto target) { + return std::ranges::find(returning, target) + != returning.end(); + })) + { + returning.push_back(block.id); + changed = true; + } + } + } + return returning.size() == function.blocks.size(); + } + + [[nodiscard]] auto + Run(const Core::Module &module, bool optimize) + -> std::optional + { + const auto encoded = Core::Wire::Encode(module); + REQUIRE(encoded); + Pipeline::Options options; + options.optimize_xpp = optimize; + options.optimize_xmm = optimize; + options.llvm.optimization = optimize ? Llvm::OptimizationLevel::Default + : Llvm::OptimizationLevel::Debug; + const auto pipeline = Pipeline::ConsumeCore(encoded.bytes, options); + REQUIRE(pipeline); + REQUIRE(pipeline.llvm); + REQUIRE(pipeline.core_prep); + REQUIRE(pipeline.core_prep->functions.size() == 1U); + REQUIRE(EveryBlockReachesReturn(pipeline.core_prep->functions.front())); + + constexpr std::string_view kSymbol = "Loops.Evaluate.1"; + Llvm::JitSession session; + const auto rejected = session.AddModule(pipeline.llvm->bitcode, + "loop-execution", + kSymbol, + Core::Type::int64()); + REQUIRE_FALSE(rejected); + const auto result = session.InvokeScalar(kSymbol, Core::Type::int64()); + REQUIRE(result); + return std::get(result.value->payload); + } + + void + CheckBothPipelines(const Core::Module &module, std::int64_t expected) + { + CHECK(Run(module, false) == expected); + CHECK(Run(module, true) == expected); + } +} // namespace + +TEST_CASE("for-loop runs its update once per iteration and then re-tests", + "[llvm][loop][execution]") +{ + for (std::int64_t limit = 0; limit < kLimits; ++limit) + { + std::int64_t expected{}; + for (std::int64_t index = 0; index < limit; ++index) + expected += index; + CAPTURE(limit); + CheckBothPipelines( + LoopModule(Core::Statement::For( + Compare(Core::Primitive::LessThan, kIndex, limit), + { Add(kTotal, Variable(kIndex)) }, + { Increment(kIndex) })), + expected); + } +} + +TEST_CASE("for-loop continue still updates and break skips the update", + "[llvm][loop][execution]") +{ + for (std::int64_t limit = 0; limit < kLimits; ++limit) + { + std::int64_t expected{}; + for (std::int64_t index = 0; index < limit; ++index) + { + if (index == 2) + continue; + if (index == 9) + break; + expected += index; + } + CAPTURE(limit); + CheckBothPipelines( + LoopModule(Core::Statement::For( + Compare(Core::Primitive::LessThan, kIndex, limit), + { When(kIndex, 2, Core::Statement::Continue()), + When(kIndex, 9, Core::Statement::Break()), + Add(kTotal, Variable(kIndex)) }, + { Increment(kIndex) })), + expected); + } +} + +TEST_CASE("for-loop observes the index value left by break and by exhaustion", + "[llvm][loop][execution]") +{ + // Returning the induction variable distinguishes "update ran after the + // last body" from "update skipped", which a sum of indices cannot. + for (std::int64_t limit = 0; limit < kLimits; ++limit) + { + std::int64_t index = 0; + for (; index < limit; ++index) + if (index == 5) + break; + auto module = LoopModule(Core::Statement::For( + Compare(Core::Primitive::LessThan, kIndex, limit), + { When(kIndex, 5, Core::Statement::Break()) }, + { Increment(kIndex) })); + module.functions.front().body.back() + = Core::Statement::Return(Variable(kIndex)); + CAPTURE(limit); + CheckBothPipelines(module, index); + } +} + +TEST_CASE("for-loop accepts a numeric condition and an empty update", + "[llvm][loop][execution]") +{ + std::int64_t expected{}; + for (std::int64_t index = 0; 3 - index; ++index) + expected += index; + CheckBothPipelines( + LoopModule(Core::Statement::For( + Core::Expression::InvokePrimitive(Core::Primitive::Subtract, + { Integer(3), Variable(kIndex) }, + Core::Type::int64()), + { Add(kTotal, Variable(kIndex)) }, + { Increment(kIndex) })), + expected); + + expected = 0; + for (std::int64_t index = 0; index < 4;) + { + expected += index; + ++index; + } + CheckBothPipelines(LoopModule(Core::Statement::For( + Compare(Core::Primitive::LessThan, kIndex, 4), + { Add(kTotal, Variable(kIndex)), Increment(kIndex) }, + {})), + expected); +} + +TEST_CASE("while-loop does not re-run statements that precede it", + "[llvm][loop][execution]") +{ + for (std::int64_t limit = 0; limit < kLimits; ++limit) + { + std::int64_t expected{}; + std::int64_t index{}; + while (index < limit) + { + if (index == 1) + { + ++index; + continue; + } + if (index == 7) + break; + expected += index; + ++index; + } + CAPTURE(limit); + CheckBothPipelines( + LoopModule(Core::Statement::While( + Compare(Core::Primitive::LessThan, kIndex, limit), + { Core::Statement::If( + Compare(Core::Primitive::Equal, kIndex, 1), + { Increment(kIndex), Core::Statement::Continue() }, + {}), + When(kIndex, 7, Core::Statement::Break()), + Add(kTotal, Variable(kIndex)), + Increment(kIndex) })), + expected); + } +} + +TEST_CASE("do-while runs its body before the first test and on continue", + "[llvm][loop][execution]") +{ + for (std::int64_t limit = 0; limit < kLimits; ++limit) + { + std::int64_t expected{}; + std::int64_t index{}; + do + { + ++index; + if (index == 2) + continue; + if (index == 5) + break; + expected += index; + } while (index < limit); + CAPTURE(limit); + CheckBothPipelines( + LoopModule(Core::Statement::DoWhile( + { Increment(kIndex), + When(kIndex, 2, Core::Statement::Continue()), + When(kIndex, 5, Core::Statement::Break()), + Add(kTotal, Variable(kIndex)) }, + Compare(Core::Primitive::LessThan, kIndex, limit))), + expected); + } +} + +TEST_CASE("nested loops transfer only within their own loop", + "[llvm][loop][execution]") +{ + for (std::int64_t limit = 0; limit < 6; ++limit) + { + std::int64_t expected{}; + for (std::int64_t index = 0; index < limit; ++index) + { + if (index == 3) + continue; + for (std::int64_t inner = 0; inner < 4; ++inner) + { + if (inner == 1) + continue; + if (inner == 3) + break; + expected += index + inner; + } + std::int64_t tail{}; + while (tail < 2) + { + expected += 100; + ++tail; + } + } + // The inner counters reuse `inner`; rebinding per outer iteration + // is expressed as an assignment so each binding is defined once. + auto reset + = Core::Statement::Assign({ kInner, Spelling(kInner) }, Integer(0)); + auto module = LoopModule(Core::Statement::For( + Compare(Core::Primitive::LessThan, kIndex, limit), + { When(kIndex, 3, Core::Statement::Continue()), + reset, + Core::Statement::For( + Compare(Core::Primitive::LessThan, kInner, 4), + { When(kInner, 1, Core::Statement::Continue()), + When(kInner, 3, Core::Statement::Break()), + Add(kTotal, Variable(kIndex)), + Add(kTotal, Variable(kInner)) }, + { Increment(kInner) }), + reset, + Core::Statement::While( + Compare(Core::Primitive::LessThan, kInner, 2), + { Add(kTotal, Integer(100)), Increment(kInner) }) }, + { Increment(kIndex) })); + auto &body = module.functions.front().body; + body.insert(body.begin() + 2, Declare(kInner)); + CAPTURE(limit); + CheckBothPipelines(module, expected); + } +} diff --git a/Compiler/Backend/LLVM/Tests/ShortCircuitExecutionTests.cpp b/Compiler/Backend/LLVM/Tests/ShortCircuitExecutionTests.cpp new file mode 100644 index 00000000..f4f2d5d1 --- /dev/null +++ b/Compiler/Backend/LLVM/Tests/ShortCircuitExecutionTests.cpp @@ -0,0 +1,318 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Backend/LLVM.hpp" +#include "Visual/XSharp/Core/IR.hpp" +#include "Visual/XSharp/Core/Wire.hpp" +#include "Visual/XSharp/Pipeline.hpp" + +// Executable short-circuit regressions. The right operand of each case is +// only well defined when the left operand guards it: a division whose +// divisor the guard excludes, or a recursive call the guard terminates. A +// pipeline that evaluates both operands traps or never returns, so these +// programs observe laziness itself rather than only the final Boolean. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Llvm = Visual::XSharp::Backend::LLVM; + namespace Pipeline = Visual::XSharp::Pipeline; + + constexpr std::uint64_t kValue = 2U; + constexpr std::uint64_t kTotal = 3U; + constexpr std::uint64_t kDown = 10U; + constexpr std::uint64_t kDownParameter = 11U; + + [[nodiscard]] auto + Integer(std::int64_t value) -> Core::Expression + { + return Core::Expression::Constant(value, Core::Type::int64()); + } + + [[nodiscard]] auto + Variable(std::uint64_t id, std::u32string spelling) -> Core::Expression + { + return Core::Expression::Variable({ id, std::move(spelling) }, + Core::Type::int64()); + } + + [[nodiscard]] auto + Value() -> Core::Expression + { + return Variable(kValue, U"value"); + } + + [[nodiscard]] auto + Binary(Core::Primitive operation, + Core::Expression left, + Core::Expression right, + Core::Type type) -> Core::Expression + { + return Core::Expression::InvokePrimitive( + operation, + { std::move(left), std::move(right) }, + std::move(type)); + } + + [[nodiscard]] auto + Compare(Core::Primitive operation, + Core::Expression left, + Core::Expression right) -> Core::Expression + { + return Binary(operation, + std::move(left), + std::move(right), + Core::Type::boolean()); + } + + [[nodiscard]] auto + Quotient(std::int64_t dividend, Core::Expression divisor) + -> Core::Expression + { + return Binary(Core::Primitive::Divide, + Integer(dividend), + std::move(divisor), + Core::Type::int64()); + } + + /// `int value = ; int total = 0; ; return total;` + [[nodiscard]] auto + Evaluate(std::int64_t input, std::vector statements) + -> Core::Function + { + std::vector body{ + Core::Statement::Bind({ { kValue, U"value" }, + Core::Type::int64(), + true, + Integer(input) }), + Core::Statement::Bind( + { { kTotal, U"total" }, Core::Type::int64(), true, Integer(0) }) + }; + body.insert(body.end(), + std::make_move_iterator(statements.begin()), + std::make_move_iterator(statements.end())); + body.push_back(Core::Statement::Return(Variable(kTotal, U"total"))); + return { { 1U, U"Evaluate" }, + {}, + Core::Type::int64(), + std::move(body) }; + } + + [[nodiscard]] auto + SetTotal(std::int64_t value) -> Core::Statement + { + return Core::Statement::Assign({ kTotal, U"total" }, Integer(value)); + } + + [[nodiscard]] auto + Run(const Core::Module &module, bool optimize) -> std::int64_t + { + const auto encoded = Core::Wire::Encode(module); + REQUIRE(encoded); + Pipeline::Options options; + options.optimize_xpp = optimize; + options.optimize_xmm = optimize; + options.llvm.optimization = optimize ? Llvm::OptimizationLevel::Default + : Llvm::OptimizationLevel::Debug; + const auto pipeline = Pipeline::ConsumeCore(encoded.bytes, options); + REQUIRE(pipeline); + REQUIRE(pipeline.llvm); + + constexpr std::string_view kSymbol = "ShortCircuit.Evaluate.1"; + Llvm::JitSession session; + const auto rejected = session.AddModule(pipeline.llvm->bitcode, + "short-circuit-execution", + kSymbol, + Core::Type::int64()); + REQUIRE_FALSE(rejected); + const auto result = session.InvokeScalar(kSymbol, Core::Type::int64()); + REQUIRE(result); + return std::get(result.value->payload); + } + + void + CheckBothPipelines(std::vector functions, + std::int64_t expected) + { + const Core::Module module{ { U"ShortCircuit" }, std::move(functions) }; + CHECK(Run(module, false) == expected); + CHECK(Run(module, true) == expected); + } +} // namespace + +TEST_CASE("logical and does not evaluate a division its left operand excludes", + "[llvm][shortcircuit][execution]") +{ + for (std::int64_t input = -3; input <= 3; ++input) + { + const std::int64_t expected = (input != 0 && 12 / input > 2) ? 1 : 0; + CAPTURE(input); + CheckBothPipelines( + { Evaluate( + input, + { Core::Statement::If( + Compare( + Core::Primitive::LogicalAnd, + Compare(Core::Primitive::NotEqual, Value(), Integer(0)), + Compare(Core::Primitive::GreaterThan, + Quotient(12, Value()), + Integer(2))), + { SetTotal(1) }, + {}) }) }, + expected); + } +} + +TEST_CASE("logical or does not evaluate a division its left operand excludes", + "[llvm][shortcircuit][execution]") +{ + for (std::int64_t input = -3; input <= 3; ++input) + { + const std::int64_t expected = (input == 0 || 12 / input < 0) ? 1 : 0; + CAPTURE(input); + CheckBothPipelines( + { Evaluate( + input, + { Core::Statement::If( + Compare( + Core::Primitive::LogicalOr, + Compare(Core::Primitive::Equal, Value(), Integer(0)), + Compare(Core::Primitive::LessThan, + Quotient(12, Value()), + Integer(0))), + { SetTotal(1) }, + {}) }) }, + expected); + } +} + +TEST_CASE("short-circuit value is usable as an ordinary Boolean binding", + "[llvm][shortcircuit][execution]") +{ + // The operator's value, not only its branch, must be the lazy result: + // bind it, then test the binding after an unrelated statement. + constexpr std::uint64_t kFlag = 4U; + for (std::int64_t input = -2; input <= 2; ++input) + { + const bool flag = input != 0 && 8 / input == 4; + const std::int64_t expected = flag ? 7 : 5; + CAPTURE(input); + CheckBothPipelines( + { Evaluate(input, + { Core::Statement::Bind( + { { kFlag, U"flag" }, + Core::Type::boolean(), + false, + Compare(Core::Primitive::LogicalAnd, + Compare(Core::Primitive::NotEqual, + Value(), + Integer(0)), + Compare(Core::Primitive::Equal, + Quotient(8, Value()), + Integer(4))) }), + SetTotal(5), + Core::Statement::If( + Core::Expression::Variable({ kFlag, U"flag" }, + Core::Type::boolean()), + { SetTotal(7) }, + {}) }) }, + expected); + } +} + +TEST_CASE("guarded recursion terminates through a short-circuit operator", + "[llvm][shortcircuit][execution]") +{ + // bool Down(int n) { return n == 0 || Down(n - 1); } + // Evaluating the call eagerly recurses below zero without bound. + const auto downType + = Core::Type::function({ Core::Type::int64() }, Core::Type::boolean()); + const auto parameter = [] { + return Variable(kDownParameter, U"n"); + }; + Core::Function down{ + { kDown, U"Down" }, + { { { kDownParameter, U"n" }, Core::Type::int64() } }, + Core::Type::boolean(), + { Core::Statement::Return(Compare( + Core::Primitive::LogicalOr, + Compare(Core::Primitive::Equal, parameter(), Integer(0)), + Core::Expression::Apply( + Core::Expression::Variable({ kDown, U"Down" }, downType), + { Binary(Core::Primitive::Subtract, + parameter(), + Integer(1), + Core::Type::int64()) }, + Core::Type::boolean()))) }, + }; + for (std::int64_t input = 0; input <= 6; ++input) + { + CAPTURE(input); + CheckBothPipelines( + { down, + Evaluate(input, + { Core::Statement::If( + Core::Expression::Apply( + Core::Expression::Variable({ kDown, U"Down" }, + downType), + { Value() }, + Core::Type::boolean()), + { SetTotal(1) }, + {}) }) }, + 1); + } +} + +TEST_CASE("short-circuit loop condition guards its own right operand", + "[llvm][shortcircuit][execution]") +{ + // while (value < limit && 100 / (limit - value) > 0) { ... } + // The division is undefined exactly when the left operand is false. + for (std::int64_t limit = 0; limit <= 6; ++limit) + { + std::int64_t expected{}; + std::int64_t value{}; + while (value < limit && 100 / (limit - value) > 0) + { + expected += value; + ++value; + } + const auto remaining = [limit] { + return Binary(Core::Primitive::Subtract, + Integer(limit), + Value(), + Core::Type::int64()); + }; + CAPTURE(limit); + CheckBothPipelines( + { Evaluate( + 0, + { Core::Statement::While( + Compare(Core::Primitive::LogicalAnd, + Compare(Core::Primitive::LessThan, + Value(), + Integer(limit)), + Compare(Core::Primitive::GreaterThan, + Quotient(100, remaining()), + Integer(0))), + { Core::Statement::Assign({ kTotal, U"total" }, + Binary(Core::Primitive::Add, + Variable(kTotal, U"total"), + Value(), + Core::Type::int64())), + Core::Statement::Assign( + { kValue, U"value" }, + Binary(Core::Primitive::Add, + Value(), + Integer(1), + Core::Type::int64())) }) }) }, + expected); + } +} diff --git a/Compiler/Cli/Arguments/Options.hpp b/Compiler/Cli/Arguments/Options.hpp index dfa91ffb..b7e63d55 100644 --- a/Compiler/Cli/Arguments/Options.hpp +++ b/Compiler/Cli/Arguments/Options.hpp @@ -116,6 +116,9 @@ struct CompilerSettings LlvmOptLevel llvmOptLevel; LlvmCompiler llvmCompiler; LlvmLto llvmLto; + + bool + operator==(const CompilerSettings &) const = default; }; struct CliOptions @@ -155,6 +158,9 @@ struct CliOptions bool llvmOptOverride; bool llvmCompilerOverride; bool llvmLtoOverride; + + bool + operator==(const CliOptions &) const = default; }; // Fully resolved values for one compiler invocation. A project evaluation can @@ -183,6 +189,9 @@ struct CliParseOutcome CliOptions options; std::optional helpCommand; std::string diagnostic; + + bool + operator==(const CliParseOutcome &) const = default; }; [[nodiscard]] CompilerSettings diff --git a/Compiler/Cli/Fuzzing/BUILD.bazel b/Compiler/Cli/Fuzzing/BUILD.bazel new file mode 100644 index 00000000..a9176242 --- /dev/null +++ b/Compiler/Cli/Fuzzing/BUILD.bazel @@ -0,0 +1,7 @@ +load("@rules_cc//cc:cc_binary.bzl", "cc_binary") + +cc_binary( + name = "cli_fuzzer", + srcs = ["CliFuzzer.cpp"], + deps = ["//Compiler/Cli/Arguments:arguments", "@llvm//:llvm"], +) diff --git a/Compiler/Cli/Fuzzing/CliFuzzer.cpp b/Compiler/Cli/Fuzzing/CliFuzzer.cpp new file mode 100644 index 00000000..b8071a8b --- /dev/null +++ b/Compiler/Cli/Fuzzing/CliFuzzer.cpp @@ -0,0 +1,54 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include + +#include "Compiler/Cli/Arguments/Options.hpp" + +extern "C" int +LLVMFuzzerTestOneInput(const std::uint8_t *data, std::size_t size) +{ + if (size > 16384U) + return 0; + std::vector arguments{ "vxs" }; + std::string word; + for (const auto byte : std::span(data, size)) + { + if (byte == 0U) + { + arguments.push_back(word); + word.clear(); + if (arguments.size() == 128U) + break; + } + else + word.push_back(static_cast(byte)); + } + if (!word.empty()) + arguments.push_back(word); + const auto original = arguments; + std::vector argv; + for (auto &argument : arguments) + argv.push_back(argument.data()); + const auto first + = ParseCommandLine(static_cast(argv.size()), argv.data()); + const auto second + = ParseCommandLine(static_cast(argv.size()), argv.data()); + // Full typed-model equality checks defaults and override bits as well as + // diagnostics. Parsing must never mutate borrowed argv storage. + if (first != second || arguments != original) + llvm::report_fatal_error( + "CLI parsing is nondeterministic or mutated argv"); + if (first.result == CliParseResult::kError && first.diagnostic.empty()) + llvm::report_fatal_error("CLI rejected input without a diagnostic"); + if (first.result == CliParseResult::kReady + && first.options.command == CliCommand::kNone) + llvm::report_fatal_error( + "CLI accepted input without selecting a command"); + return 0; +} diff --git a/Compiler/Core/CorePrep/Prepare.cpp b/Compiler/Core/CorePrep/Prepare.cpp index f174b4c9..d5664ee8 100644 --- a/Compiler/Core/CorePrep/Prepare.cpp +++ b/Compiler/Core/CorePrep/Prepare.cpp @@ -25,29 +25,6 @@ namespace Visual::XSharp::Core::CorePrep loopTargets; }; - struct Atomized final - { - std::vector prefix; - Prepared::Atom atom; - State state; - }; - - struct OperationResult final - { - std::vector prefix; - Prepared::Operation operation{ Prepared::Operation::Copy }; - std::vector operands; - SymbolName closureFunction{}; - std::vector captures; - State state; - }; - - struct BlocksResult final - { - std::vector blocks; - State state; - }; - [[nodiscard]] auto LowerLiteral(const Expression &expression) -> Prepared::Atom { @@ -118,27 +95,95 @@ namespace Visual::XSharp::Core::CorePrep std::abort(); } + /** + * @brief The block receiving instructions plus every finished block. + * + * Expression atomization is not confined to one block: a + * short-circuit operator ends the current block with a branch and + * continues in a join block. Threading one cursor through both + * expressions and statements lets either of them close and reopen + * blocks without a second lowering path. + */ + struct Cursor final + { + State state; + Prepared::BlockId block{}; + std::vector instructions; + std::vector closed; + /// False after a return, break or continue ended the region. + bool open{ true }; + + void + Emit(Prepared::Instruction instruction) + { + instructions.push_back(std::move(instruction)); + } + + /// End the current block; no block is open until Open is called. + void + Close(Prepared::Terminator terminator) + { + closed.push_back( + { block, std::move(instructions), std::move(terminator) }); + instructions.clear(); + open = false; + } + + void + Open(const Prepared::BlockId id) + { + block = id; + instructions.clear(); + open = true; + } + + void + Jump(const Prepared::BlockId target) + { + Close({ Prepared::Terminator::Kind::Jump, {}, target, 0U }); + } + + void + Branch(Prepared::Atom condition, + const Prepared::BlockId whenTrue, + const Prepared::BlockId whenFalse) + { + Close({ Prepared::Terminator::Kind::Branch, + std::move(condition), + whenTrue, + whenFalse }); + } + + [[nodiscard]] auto + Temporary(std::u32string prefix) -> SymbolName + { + const auto id = state.nextTemporary++; + const auto digits = std::to_string(id); + prefix.append(digits.begin(), digits.end()); + return SymbolName{ id, std::move(prefix) }; + } + }; + + struct OperationResult final + { + Prepared::Operation operation{ Prepared::Operation::Copy }; + std::vector operands; + SymbolName closureFunction{}; + std::vector captures; + }; + [[nodiscard]] auto - Atomize(State state, const Expression &expression) -> Atomized; + Atomize(Cursor &cursor, const Expression &expression) -> Prepared::Atom; [[nodiscard]] auto - AtomizeMany(State state, const std::vector &expressions) - -> std::pair, - std::pair, State>> + AtomizeMany(Cursor &cursor, const std::vector &expressions) + -> std::vector { - std::vector prefix; std::vector atoms; atoms.reserve(expressions.size()); for (const auto &expression : expressions) - { - auto atomized = Atomize(state, expression); - prefix.insert(prefix.end(), - std::make_move_iterator(atomized.prefix.begin()), - std::make_move_iterator(atomized.prefix.end())); - atoms.push_back(std::move(atomized.atom)); - state = atomized.state; - } - return { std::move(prefix), { std::move(atoms), state } }; + atoms.push_back(Atomize(cursor, expression)); + return atoms; } [[nodiscard]] auto @@ -152,20 +197,18 @@ namespace Visual::XSharp::Core::CorePrep type); } + /// Canonicalize a numeric truth value to `value != 0` in the open + /// block; a Boolean atom is returned unchanged. [[nodiscard]] auto - Booleanize(State state, Prepared::Atom atom) -> Atomized + Booleanize(Cursor &cursor, Prepared::Atom atom) -> Prepared::Atom { if (atom.type == Type::boolean()) - return { {}, std::move(atom), std::move(state) }; - const auto id = state.nextTemporary++; - const auto digits = std::to_string(id); - std::u32string spelling = U"$condition"; - spelling.append(digits.begin(), digits.end()); - auto temporary = SymbolName{ id, std::move(spelling) }; + return atom; + auto temporary = cursor.Temporary(U"$condition"); std::vector operands; operands.push_back(std::move(atom)); operands.push_back(ZeroForBooleanContext(operands.front().type)); - Prepared::Instruction comparison{ + cursor.Emit(Prepared::Instruction{ Prepared::Instruction::Kind::Bind, temporary, Type::boolean(), @@ -174,41 +217,78 @@ namespace Visual::XSharp::Core::CorePrep std::move(operands), {}, {}, - }; - return { - { std::move(comparison) }, - Prepared::Atom::variable(std::move(temporary), Type::boolean()), - std::move(state), - }; + }); + return Prepared::Atom::variable(std::move(temporary), + Type::boolean()); } [[nodiscard]] auto - BooleanizeMany(State state, std::vector atoms) - -> std::pair, - std::pair, State>> + IsShortCircuit(const Expression &expression) -> bool { - std::vector prefix; - std::vector booleans; - booleans.reserve(atoms.size()); - for (auto &atom : atoms) - { - auto boolean = Booleanize(std::move(state), std::move(atom)); - prefix.insert(prefix.end(), - std::make_move_iterator(boolean.prefix.begin()), - std::make_move_iterator(boolean.prefix.end())); - booleans.push_back(std::move(boolean.atom)); - state = std::move(boolean.state); - } - return { std::move(prefix), - { std::move(booleans), std::move(state) } }; + return expression.kind == Expression::Kind::Primitive + && (expression.primitive == Primitive::LogicalAnd + || expression.primitive == Primitive::LogicalOr) + && expression.operands.size() == 2U; } - struct CapturesResult final + /** + * @brief Lower `&&` and `||` as control flow, not as an eager operator. + * + * The right operand is evaluated only on the path that needs it, so + * its calls, traps and non-termination stay conditional exactly as + * the source wrote them. The result slot is initialized with the + * short-circuit value before the branch and overwritten only on the + * path that evaluates the right operand; every predecessor of the + * join therefore carries an initialized Boolean without a phi node + * in CorePrep's storage-oriented form. This matches the Haskell + * CorePrep lowering. + */ + [[nodiscard]] auto + AtomizeShortCircuit(Cursor &cursor, const Expression &expression) + -> Prepared::Atom { - std::vector prefix; - std::vector captures; - State state; - }; + const auto isOr = expression.primitive == Primitive::LogicalOr; + auto condition + = Booleanize(cursor, + Atomize(cursor, expression.operands.front())); + auto result = cursor.Temporary(U"$shortcircuit"); + cursor.Emit(Prepared::Instruction{ + Prepared::Instruction::Kind::Bind, + result, + Type::boolean(), + true, + Prepared::Operation::Copy, + { Prepared::Atom::constant(Prepared::Literal{ isOr }, + Type::boolean()) }, + {}, + {}, + }); + const auto rightId = cursor.state.nextBlock; + const auto joinId = rightId + 1U; + cursor.state.nextBlock = joinId + 1U; + if (isOr) + cursor.Branch(std::move(condition), joinId, rightId); + else + cursor.Branch(std::move(condition), rightId, joinId); + + cursor.Open(rightId); + auto right + = Booleanize(cursor, + Atomize(cursor, expression.operands.back())); + cursor.Emit(Prepared::Instruction{ + Prepared::Instruction::Kind::Assign, + result, + Type::boolean(), + false, + Prepared::Operation::Copy, + { std::move(right) }, + {}, + {}, + }); + cursor.Jump(joinId); + cursor.Open(joinId); + return Prepared::Atom::variable(std::move(result), Type::boolean()); + } struct PreparedFunctionResult final { @@ -304,10 +384,9 @@ namespace Visual::XSharp::Core::CorePrep } [[nodiscard]] auto - AtomizeCaptures(State state, const std::vector &captures) - -> CapturesResult + AtomizeCaptures(Cursor &cursor, const std::vector &captures) + -> std::vector { - std::vector prefix; std::vector prepared; prepared.reserve(captures.size()); for (const auto &capture : captures) @@ -316,84 +395,60 @@ namespace Visual::XSharp::Core::CorePrep // for direct native API callers as well. if (!capture.value) std::abort(); - auto value = Atomize(state, *capture.value); - prefix.insert(prefix.end(), - std::make_move_iterator(value.prefix.begin()), - std::make_move_iterator(value.prefix.end())); prepared.push_back(Prepared::Capture{ capture.mode, capture.symbol, capture.type, - std::move(value.atom), + Atomize(cursor, *capture.value), }); - state = std::move(value.state); } - return { std::move(prefix), std::move(prepared), std::move(state) }; + return prepared; } [[nodiscard]] auto - AtomizeOperation(State state, const Expression &expression) + AtomizeOperation(Cursor &cursor, const Expression &expression) -> OperationResult { - if (expression.kind == Expression::Kind::Let) - { - auto atomized = Atomize(std::move(state), expression); - return { std::move(atomized.prefix), - Prepared::Operation::Copy, - { std::move(atomized.atom) }, + if (expression.kind == Expression::Kind::Let + || IsShortCircuit(expression)) + return { Prepared::Operation::Copy, + { Atomize(cursor, expression) }, {}, - {}, - std::move(atomized.state) }; - } + {} }; if (expression.kind == Expression::Kind::Variable) - return { {}, - Prepared::Operation::Copy, + return { Prepared::Operation::Copy, { Prepared::Atom::variable(expression.symbol, expression.type) }, {}, - {}, - state }; + {} }; if (expression.kind == Expression::Kind::Literal) - return { {}, - Prepared::Operation::Copy, + return { Prepared::Operation::Copy, { LowerLiteral(expression) }, {}, - {}, - state }; + {} }; if (expression.kind == Expression::Kind::Apply) { - auto callee = Atomize(state, *expression.callee); - auto arguments = AtomizeMany(callee.state, expression.operands); - callee.prefix.insert( - callee.prefix.end(), - std::make_move_iterator(arguments.first.begin()), - std::make_move_iterator(arguments.first.end())); std::vector operands; - operands.reserve(arguments.second.first.size() + 1U); - operands.push_back(std::move(callee.atom)); - operands.insert( - operands.end(), - std::make_move_iterator(arguments.second.first.begin()), - std::make_move_iterator(arguments.second.first.end())); - return { std::move(callee.prefix), - Prepared::Operation::Call, + operands.reserve(expression.operands.size() + 1U); + operands.push_back(Atomize(cursor, *expression.callee)); + for (const auto &argument : expression.operands) + operands.push_back(Atomize(cursor, argument)); + return { Prepared::Operation::Call, std::move(operands), {}, - {}, - arguments.second.second }; + {} }; } if (expression.kind == Expression::Kind::Closure) { if (!expression.closureBody) std::abort(); - const auto closureId = state.nextFunction++; + const auto closureId = cursor.state.nextFunction++; const auto digits = std::to_string(closureId); std::u32string spelling = U"$closure"; spelling.append(digits.begin(), digits.end()); SymbolName closureName{ closureId, std::move(spelling) }; - auto captures - = AtomizeCaptures(std::move(state), expression.captures); + auto captures = AtomizeCaptures(cursor, expression.captures); std::vector parameters; parameters.reserve(expression.captures.size() + expression.closureParameters.size()); @@ -402,494 +457,322 @@ namespace Visual::XSharp::Core::CorePrep Parameter{ capture.symbol, capture.type }); for (const auto &[symbol, type] : expression.closureParameters) parameters.push_back(Parameter{ symbol, type }); - captures.state.pendingFunctions.push_back(Function{ + cursor.state.pendingFunctions.push_back(Function{ closureName, std::move(parameters), expression.closureReturnType, *expression.closureBody, }); - return { - std::move(captures.prefix), - Prepared::Operation::MakeClosure, - {}, - std::move(closureName), - std::move(captures.captures), - std::move(captures.state), - }; + return { Prepared::Operation::MakeClosure, + {}, + std::move(closureName), + std::move(captures) }; } - auto arguments = AtomizeMany(state, expression.operands); + auto operands = AtomizeMany(cursor, expression.operands); if (expression.primitive == Primitive::LogicalAnd || expression.primitive == Primitive::LogicalOr || expression.primitive == Primitive::LogicalNot) - { - auto booleans - = BooleanizeMany(std::move(arguments.second.second), - std::move(arguments.second.first)); - arguments.first.insert( - arguments.first.end(), - std::make_move_iterator(booleans.first.begin()), - std::make_move_iterator(booleans.first.end())); - return { std::move(arguments.first), - LowerPrimitive(expression.primitive), - std::move(booleans.second.first), - {}, - {}, - std::move(booleans.second.second) }; - } - return { std::move(arguments.first), - LowerPrimitive(expression.primitive), - std::move(arguments.second.first), - {}, + for (auto &operand : operands) + operand = Booleanize(cursor, std::move(operand)); + return { LowerPrimitive(expression.primitive), + std::move(operands), {}, - arguments.second.second }; + {} }; } [[nodiscard]] auto - Atomize(State state, const Expression &expression) -> Atomized + Atomize(Cursor &cursor, const Expression &expression) -> Prepared::Atom { if (expression.kind == Expression::Kind::Variable) - return { {}, - Prepared::Atom::variable(expression.symbol, - expression.type), - state }; + return Prepared::Atom::variable(expression.symbol, + expression.type); if (expression.kind == Expression::Kind::Literal) - return { {}, LowerLiteral(expression), state }; + return LowerLiteral(expression); if (expression.kind == Expression::Kind::Let) { if (!expression.letValue || !expression.letBody) std::abort(); - auto value = Atomize(std::move(state), *expression.letValue); - value.prefix.push_back( + auto value = Atomize(cursor, *expression.letValue); + cursor.Emit( Prepared::Instruction{ Prepared::Instruction::Kind::Bind, expression.letSymbol, expression.letType, false, Prepared::Operation::Copy, - { std::move(value.atom) }, + { std::move(value) }, {}, {} }); - auto body - = Atomize(std::move(value.state), *expression.letBody); - value.prefix.insert( - value.prefix.end(), - std::make_move_iterator(body.prefix.begin()), - std::make_move_iterator(body.prefix.end())); - return { std::move(value.prefix), - std::move(body.atom), - std::move(body.state) }; + return Atomize(cursor, *expression.letBody); } - - auto operation = AtomizeOperation(state, expression); - const auto id = operation.state.nextTemporary++; - const auto digits = std::to_string(id); - std::u32string spelling = U"$coreprep"; - spelling.append(digits.begin(), digits.end()); - auto temporary = SymbolName{ id, std::move(spelling) }; - Prepared::Instruction binding{ Prepared::Instruction::Kind::Bind, - temporary, - expression.type, - false, - operation.operation, - std::move(operation.operands), - std::move(operation.closureFunction), - std::move(operation.captures) }; - operation.prefix.push_back(std::move(binding)); - return { std::move(operation.prefix), - Prepared::Atom::variable(std::move(temporary), - expression.type), - operation.state }; + if (IsShortCircuit(expression)) + return AtomizeShortCircuit(cursor, expression); + + auto operation = AtomizeOperation(cursor, expression); + auto temporary = cursor.Temporary(U"$coreprep"); + cursor.Emit( + Prepared::Instruction{ Prepared::Instruction::Kind::Bind, + temporary, + expression.type, + false, + operation.operation, + std::move(operation.operands), + std::move(operation.closureFunction), + std::move(operation.captures) }); + return Prepared::Atom::variable(std::move(temporary), + expression.type); } - [[nodiscard]] auto - PrepareStatements(State state, - Prepared::BlockId blockId, - std::vector instructions, - const std::vector &statements, - std::size_t start = 0U) -> BlocksResult; - - /** - * @brief Replace open-region fallthrough sentinels with a real edge. - * - * `PrepareStatements` uses `Unreachable` for the still-open tail of a - * nested region. Explicit `Return` and `Jump` terminators are closed - * paths and are deliberately not rewritten. - */ void - ConnectFallthrough(std::vector &blocks, - const Prepared::BlockId target) - { - for (auto &block : blocks) - { - if (block.terminator.kind - == Prepared::Terminator::Kind::Unreachable) - { - block.terminator.kind = Prepared::Terminator::Kind::Jump; - block.terminator.true_target = target; - } - } - } + PrepareStatements(Cursor &cursor, + const std::vector &statements); /** - * @brief Prepare a loop region with a scoped transfer-target stack. + * @brief Prepare one loop region with a scoped transfer-target pair. + * + * The region starts in a fresh block. While it is built, `break` + * jumps to breakTarget and `continue` to continueTarget; the + * enclosing pair is restored afterwards, so an inner transfer + * cannot target an outer loop. A region that is still open at its + * end jumps to fallthroughTarget. * - * Nested regions inherit this loop's targets while they are built, - * then restore the enclosing stack. Thus an inner `break` or - * `continue` cannot accidentally target an outer loop. Only blocks - * still marked as fallthrough are connected to the continuation; - * explicit return and jump terminators remain untouched. + * The fallthrough successor is a separate argument because it is + * not always the `continue` target: a for-loop update region is + * entered by `continue` but must fall through to the condition. + * Reusing the continue target there makes the region branch to + * itself and never re-test the loop condition. */ - [[nodiscard]] auto - PrepareLoopRegion(State state, + void + PrepareLoopRegion(Cursor &cursor, const Prepared::BlockId startId, const Prepared::BlockId breakTarget, const Prepared::BlockId continueTarget, + const Prepared::BlockId fallthroughTarget, const std::vector &statements) - -> BlocksResult { - auto enclosingTargets = state.loopTargets; - state.loopTargets.emplace_back(breakTarget, continueTarget); - auto result = PrepareStatements(state, startId, {}, statements); - ConnectFallthrough(result.blocks, continueTarget); - result.state.loopTargets = std::move(enclosingTargets); - return result; + cursor.state.loopTargets.emplace_back(breakTarget, continueTarget); + cursor.Open(startId); + PrepareStatements(cursor, statements); + if (cursor.open) + cursor.Jump(fallthroughTarget); + cursor.state.loopTargets.pop_back(); } + /// Evaluate a loop or branch condition in the open block and return + /// its Boolean atom. Short-circuit operands may leave a different + /// block open than the one the condition started in. [[nodiscard]] auto - PrepareBranch(State state, - Prepared::BlockId blockId, - Prepared::BlockId joinId, - const std::vector &statements) -> BlocksResult + PrepareCondition(Cursor &cursor, const Expression &condition) + -> Prepared::Atom { - auto result = PrepareStatements(state, blockId, {}, statements); - for (auto &block : result.blocks) - { - if (block.terminator.kind - == Prepared::Terminator::Kind::Unreachable) - { - block.terminator.kind = Prepared::Terminator::Kind::Jump; - block.terminator.true_target = joinId; - } - } - return result; + return Booleanize(cursor, Atomize(cursor, condition)); } - [[nodiscard]] auto - PrepareStatements(State state, - Prepared::BlockId blockId, - std::vector instructions, - const std::vector &statements, - std::size_t start) -> BlocksResult + void + PrepareBranchRegion(Cursor &cursor, + const Prepared::BlockId startId, + const Prepared::BlockId joinId, + const std::vector &statements) { - for (std::size_t index = start; index < statements.size(); ++index) - { - const auto &statement = statements[index]; - if (statement.kind == Statement::Kind::Bind) - { - auto operation - = AtomizeOperation(state, statement.binding.value); - instructions.insert( - instructions.end(), - std::make_move_iterator(operation.prefix.begin()), - std::make_move_iterator(operation.prefix.end())); - instructions.push_back(Prepared::Instruction{ - Prepared::Instruction::Kind::Bind, - statement.binding.symbol, - statement.binding.type, - statement.binding.mutableBinding, - operation.operation, - std::move(operation.operands), - std::move(operation.closureFunction), - std::move(operation.captures) }); - state = operation.state; - continue; - } - if (statement.kind == Statement::Kind::Assign) - { - auto value = Atomize(state, statement.expression); - instructions.insert( - instructions.end(), - std::make_move_iterator(value.prefix.begin()), - std::make_move_iterator(value.prefix.end())); - instructions.push_back(Prepared::Instruction{ - Prepared::Instruction::Kind::Assign, - statement.destination, - statement.expression.type, - false, - Prepared::Operation::Copy, - { std::move(value.atom) }, - {}, - {} }); - state = value.state; - continue; - } - if (statement.kind == Statement::Kind::Evaluate) - { - auto operation - = AtomizeOperation(state, statement.expression); - instructions.insert( - instructions.end(), - std::make_move_iterator(operation.prefix.begin()), - std::make_move_iterator(operation.prefix.end())); - instructions.push_back(Prepared::Instruction{ - Prepared::Instruction::Kind::Evaluate, - {}, - statement.expression.type, - false, - operation.operation, - std::move(operation.operands), - std::move(operation.closureFunction), - std::move(operation.captures) }); - state = operation.state; - continue; - } - if (statement.kind == Statement::Kind::Return) - { - auto value = Atomize(state, statement.expression); - instructions.insert( - instructions.end(), - std::make_move_iterator(value.prefix.begin()), - std::make_move_iterator(value.prefix.end())); - return { { { blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Return, - std::move(value.atom), - 0U, - 0U } } }, - value.state }; - } + cursor.Open(startId); + PrepareStatements(cursor, statements); + if (cursor.open) + cursor.Jump(joinId); + } - if (statement.kind == Statement::Kind::Break - || statement.kind == Statement::Kind::Continue) + /** + * @brief Append structured statements to the cursor's open block. + * + * On return the cursor is either still open, meaning control falls + * through to whatever the caller places next, or closed by a + * return, break or continue, after which the remaining statements + * of the region are unreachable and are not lowered. + */ + void + PrepareStatements(Cursor &cursor, + const std::vector &statements) + { + for (const auto &statement : statements) + { + switch (statement.kind) { - if (state.loopTargets.empty()) + case Statement::Kind::Bind: + { + auto operation + = AtomizeOperation(cursor, statement.binding.value); + cursor.Emit(Prepared::Instruction{ + Prepared::Instruction::Kind::Bind, + statement.binding.symbol, + statement.binding.type, + statement.binding.mutableBinding, + operation.operation, + std::move(operation.operands), + std::move(operation.closureFunction), + std::move(operation.captures) }); + break; + } + case Statement::Kind::Assign: { - return { { { blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Unreachable, - {}, + auto value = Atomize(cursor, statement.expression); + cursor.Emit(Prepared::Instruction{ + Prepared::Instruction::Kind::Assign, + statement.destination, + statement.expression.type, + false, + Prepared::Operation::Copy, + { std::move(value) }, + {}, + {} }); + break; + } + case Statement::Kind::Evaluate: + { + auto operation + = AtomizeOperation(cursor, statement.expression); + cursor.Emit(Prepared::Instruction{ + Prepared::Instruction::Kind::Evaluate, + {}, + statement.expression.type, + false, + operation.operation, + std::move(operation.operands), + std::move(operation.closureFunction), + std::move(operation.captures) }); + break; + } + case Statement::Kind::Return: + { + auto value = Atomize(cursor, statement.expression); + cursor.Close({ Prepared::Terminator::Kind::Return, + std::move(value), 0U, - 0U } } }, - state }; + 0U }); + return; + } + case Statement::Kind::Break: + case Statement::Kind::Continue: + { + // Core verification rejects a transfer outside a + // loop; stay total for direct native API callers. + if (cursor.state.loopTargets.empty()) + { + cursor.Close( + { Prepared::Terminator::Kind::Unreachable, + {}, + 0U, + 0U }); + return; + } + const auto targets = cursor.state.loopTargets.back(); + cursor.Jump(statement.kind == Statement::Kind::Break + ? targets.first + : targets.second); + return; + } + case Statement::Kind::While: + { + // The condition owns a dedicated header block. + // Folding it into the incoming block would make the + // back-edge re-execute every straight-line statement + // that precedes the loop, including the initializers + // it tests. + const auto conditionId = cursor.state.nextBlock; + const auto bodyId = conditionId + 1U; + const auto exitId = bodyId + 1U; + cursor.state.nextBlock = exitId + 1U; + + cursor.Jump(conditionId); + cursor.Open(conditionId); + cursor.Branch( + PrepareCondition(cursor, statement.expression), + bodyId, + exitId); + PrepareLoopRegion(cursor, + bodyId, + exitId, + conditionId, + conditionId, + statement.loopBody); + cursor.Open(exitId); + break; + } + case Statement::Kind::DoWhile: + { + const auto bodyId = cursor.state.nextBlock; + const auto conditionId = bodyId + 1U; + const auto exitId = conditionId + 1U; + cursor.state.nextBlock = exitId + 1U; + + cursor.Jump(bodyId); + PrepareLoopRegion(cursor, + bodyId, + exitId, + conditionId, + conditionId, + statement.loopBody); + cursor.Open(conditionId); + cursor.Branch( + PrepareCondition(cursor, statement.expression), + bodyId, + exitId); + cursor.Open(exitId); + break; + } + case Statement::Kind::For: + { + const auto conditionId = cursor.state.nextBlock; + const auto bodyId = conditionId + 1U; + const auto updateId = bodyId + 1U; + const auto exitId = updateId + 1U; + cursor.state.nextBlock = exitId + 1U; + + cursor.Jump(conditionId); + cursor.Open(conditionId); + // A numeric condition's canonicalizing comparison is + // part of the header and is re-evaluated each pass. + cursor.Branch( + PrepareCondition(cursor, statement.expression), + bodyId, + exitId); + PrepareLoopRegion(cursor, + bodyId, + exitId, + updateId, + updateId, + statement.loopBody); + // `continue` enters the update region; the region + // itself then returns to the condition, never to its + // own entry. + PrepareLoopRegion(cursor, + updateId, + exitId, + updateId, + conditionId, + statement.loopUpdate); + cursor.Open(exitId); + break; + } + case Statement::Kind::If: + { + auto condition + = PrepareCondition(cursor, statement.expression); + const auto trueId = cursor.state.nextBlock; + const auto falseId = trueId + 1U; + const auto joinId = falseId + 1U; + cursor.state.nextBlock = joinId + 1U; + cursor.Branch(std::move(condition), trueId, falseId); + PrepareBranchRegion(cursor, + trueId, + joinId, + statement.trueBranch); + PrepareBranchRegion(cursor, + falseId, + joinId, + statement.falseBranch); + cursor.Open(joinId); + break; } - const auto targets = state.loopTargets.back(); - const auto target = statement.kind == Statement::Kind::Break - ? targets.first - : targets.second; - return { { { blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Jump, - {}, - target, - 0U } } }, - state }; - } - - if (statement.kind == Statement::Kind::While) - { - auto condition = Atomize(state, statement.expression); - auto boolean = Booleanize(std::move(condition.state), - std::move(condition.atom)); - condition.prefix.insert( - condition.prefix.end(), - std::make_move_iterator(boolean.prefix.begin()), - std::make_move_iterator(boolean.prefix.end())); - instructions.insert( - instructions.end(), - std::make_move_iterator(condition.prefix.begin()), - std::make_move_iterator(condition.prefix.end())); - - const auto bodyId = boolean.state.nextBlock; - const auto exitId = bodyId + 1U; - boolean.state.nextBlock = exitId + 1U; - auto body = PrepareLoopRegion(boolean.state, - bodyId, - exitId, - blockId, - statement.loopBody); - auto tail = PrepareStatements(body.state, - exitId, - {}, - statements, - index + 1U); - std::vector blocks; - blocks.push_back({ blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Branch, - std::move(boolean.atom), - bodyId, - exitId } }); - blocks.insert(blocks.end(), - std::make_move_iterator(body.blocks.begin()), - std::make_move_iterator(body.blocks.end())); - blocks.insert(blocks.end(), - std::make_move_iterator(tail.blocks.begin()), - std::make_move_iterator(tail.blocks.end())); - return { std::move(blocks), std::move(tail.state) }; - } - - if (statement.kind == Statement::Kind::DoWhile) - { - const auto bodyId = state.nextBlock; - const auto conditionId = bodyId + 1U; - const auto exitId = conditionId + 1U; - state.nextBlock = exitId + 1U; - auto body = PrepareLoopRegion(state, - bodyId, - exitId, - conditionId, - statement.loopBody); - auto condition = Atomize(body.state, statement.expression); - auto boolean = Booleanize(std::move(condition.state), - std::move(condition.atom)); - condition.prefix.insert( - condition.prefix.end(), - std::make_move_iterator(boolean.prefix.begin()), - std::make_move_iterator(boolean.prefix.end())); - auto tail = PrepareStatements(boolean.state, - exitId, - {}, - statements, - index + 1U); - - std::vector blocks; - blocks.push_back({ blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Jump, - {}, - bodyId, - 0U } }); - blocks.insert(blocks.end(), - std::make_move_iterator(body.blocks.begin()), - std::make_move_iterator(body.blocks.end())); - blocks.push_back({ conditionId, - std::move(condition.prefix), - { Prepared::Terminator::Kind::Branch, - std::move(boolean.atom), - bodyId, - exitId } }); - blocks.insert(blocks.end(), - std::make_move_iterator(tail.blocks.begin()), - std::make_move_iterator(tail.blocks.end())); - return { std::move(blocks), std::move(tail.state) }; - } - - if (statement.kind == Statement::Kind::For) - { - const auto conditionId = state.nextBlock; - const auto bodyId = conditionId + 1U; - const auto updateId = bodyId + 1U; - const auto exitId = updateId + 1U; - state.nextBlock = exitId + 1U; - - auto condition = Atomize(state, statement.expression); - auto boolean = Booleanize(std::move(condition.state), - std::move(condition.atom)); - auto body = PrepareLoopRegion(boolean.state, - bodyId, - exitId, - updateId, - statement.loopBody); - auto update = PrepareLoopRegion(body.state, - updateId, - exitId, - updateId, - statement.loopUpdate); - auto tail = PrepareStatements(update.state, - exitId, - {}, - statements, - index + 1U); - - std::vector blocks; - blocks.push_back({ blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Jump, - {}, - conditionId, - 0U } }); - blocks.push_back({ conditionId, - std::move(condition.prefix), - { Prepared::Terminator::Kind::Branch, - std::move(boolean.atom), - bodyId, - exitId } }); - blocks.insert(blocks.end(), - std::make_move_iterator(body.blocks.begin()), - std::make_move_iterator(body.blocks.end())); - blocks.insert( - blocks.end(), - std::make_move_iterator(update.blocks.begin()), - std::make_move_iterator(update.blocks.end())); - blocks.insert(blocks.end(), - std::make_move_iterator(tail.blocks.begin()), - std::make_move_iterator(tail.blocks.end())); - return { std::move(blocks), std::move(tail.state) }; - } - - if (statement.kind != Statement::Kind::If) - { - return { { { blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Unreachable, - {}, - 0U, - 0U } } }, - state }; } - auto condition = Atomize(state, statement.expression); - auto boolean = Booleanize(std::move(condition.state), - std::move(condition.atom)); - instructions.insert( - instructions.end(), - std::make_move_iterator(condition.prefix.begin()), - std::make_move_iterator(condition.prefix.end())); - instructions.insert( - instructions.end(), - std::make_move_iterator(boolean.prefix.begin()), - std::make_move_iterator(boolean.prefix.end())); - const auto trueId = boolean.state.nextBlock; - const auto falseId = trueId + 1U; - const auto joinId = falseId + 1U; - boolean.state.nextBlock = joinId + 1U; - auto trueBlocks = PrepareBranch(boolean.state, - trueId, - joinId, - statement.trueBranch); - auto falseBlocks = PrepareBranch(trueBlocks.state, - falseId, - joinId, - statement.falseBranch); - auto tail = PrepareStatements(falseBlocks.state, - joinId, - {}, - statements, - index + 1U); - std::vector blocks; - blocks.push_back({ blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Branch, - std::move(boolean.atom), - trueId, - falseId } }); - blocks.insert( - blocks.end(), - std::make_move_iterator(trueBlocks.blocks.begin()), - std::make_move_iterator(trueBlocks.blocks.end())); - blocks.insert( - blocks.end(), - std::make_move_iterator(falseBlocks.blocks.begin()), - std::make_move_iterator(falseBlocks.blocks.end())); - blocks.insert(blocks.end(), - std::make_move_iterator(tail.blocks.begin()), - std::make_move_iterator(tail.blocks.end())); - return { std::move(blocks), tail.state }; } - return { - { { blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Unreachable, {}, 0U, 0U } } }, - state - }; } [[nodiscard]] auto @@ -903,19 +786,25 @@ namespace Visual::XSharp::Core::CorePrep { parameters.push_back({ parameter.symbol, parameter.type }); } - auto body = PrepareStatements( - State{ highest + 1U, 1U, nextFunction, {}, {} }, - 0U, - {}, - function.body); + Cursor cursor{ State{ highest + 1U, 1U, nextFunction, {}, {} }, + 0U, + {}, + {}, + true }; + PrepareStatements(cursor, function.body); + // Core verification proves every path returns. A body that still + // falls off its end is marked instead of given an invented value. + if (cursor.open) + cursor.Close( + { Prepared::Terminator::Kind::Unreachable, {}, 0U, 0U }); return { Prepared::Function{ function.symbol, std::move(parameters), function.returnType, 0U, - std::move(body.blocks) }, - std::move(body.state.pendingFunctions), - body.state.nextFunction, + std::move(cursor.closed) }, + std::move(cursor.state.pendingFunctions), + cursor.state.nextFunction, }; } } // namespace diff --git a/Compiler/Core/Tests/BUILD.bazel b/Compiler/Core/Tests/BUILD.bazel index 177c4bc8..785839bd 100644 --- a/Compiler/Core/Tests/BUILD.bazel +++ b/Compiler/Core/Tests/BUILD.bazel @@ -16,6 +16,8 @@ cc_binary( name = "core_pipeline_tests", srcs = [ "CorePipelineTests.cpp", + "LoopLoweringTests.cpp", + "ShortCircuitLoweringTests.cpp", "TemplateTests.cpp", ], deps = [ diff --git a/Compiler/Core/Tests/CorePipelineTests.cpp b/Compiler/Core/Tests/CorePipelineTests.cpp index 89693ffd..e6c7edcc 100644 --- a/Compiler/Core/Tests/CorePipelineTests.cpp +++ b/Compiler/Core/Tests/CorePipelineTests.cpp @@ -681,15 +681,37 @@ TEST_CASE( REQUIRE(Core::Verify(module).empty()); const auto prepared = Core::CorePrep::Prepare(module); REQUIRE(visual_xsharp::core::verify(prepared).empty()); - const auto &instructions - = prepared.functions.front().blocks.front().instructions; - REQUIRE(instructions.size() == 3U); - CHECK(instructions.at(0).operation + // Each operand is canonicalized with its own typed `!= 0` comparison. + // The conjunction itself is control flow: the left comparison and the + // result slot live in the entry block, and the right comparison runs + // only in the block the true edge reaches. + const auto &blocks = prepared.functions.front().blocks; + REQUIRE(blocks.size() == 3U); + const auto &entry = blocks.front(); + REQUIRE(entry.instructions.size() == 2U); + CHECK(entry.instructions.at(0).operation == visual_xsharp::core::Operation::NotEqual); - CHECK(instructions.at(1).operation + CHECK(entry.instructions.at(0).operands.front().type + == Core::Type::int64()); + CHECK(entry.instructions.at(1).operation + == visual_xsharp::core::Operation::Copy); + REQUIRE(entry.terminator.kind + == visual_xsharp::core::Terminator::Kind::Branch); + const auto right = std::ranges::find(blocks, + entry.terminator.true_target, + &visual_xsharp::core::Block::id); + REQUIRE(right != blocks.end()); + REQUIRE(right->instructions.size() == 2U); + CHECK(right->instructions.at(0).operation == visual_xsharp::core::Operation::NotEqual); - CHECK(instructions.at(2).operation - == visual_xsharp::core::Operation::LogicalAnd); + CHECK(right->instructions.at(0).operands.front().type + == Core::Type::float32()); + CHECK(right->instructions.at(1).kind + == visual_xsharp::core::Instruction::Kind::Assign); + for (const auto &block : blocks) + for (const auto &instruction : block.instructions) + CHECK(instruction.operation + != visual_xsharp::core::Operation::LogicalAnd); const auto result = Visual::XSharp::Pipeline::ConsumeCore( Core::Wire::Encode(module).bytes); REQUIRE(result); diff --git a/Compiler/Core/Tests/LoopLoweringTests.cpp b/Compiler/Core/Tests/LoopLoweringTests.cpp new file mode 100644 index 00000000..cd84a2cd --- /dev/null +++ b/Compiler/Core/Tests/LoopLoweringTests.cpp @@ -0,0 +1,453 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Core/CorePrep/Prepare.hpp" +#include "Visual/XSharp/Core/CorePrep/Verifier.hpp" +#include "Visual/XSharp/Core/Verifier.hpp" + +// These tests pin the control-flow graph the native Core-to-CorePrep adapter +// builds for structured loops. They assert exact edges instead of "some +// back-edge exists": a for-loop whose update region branched to itself used +// to satisfy every weaker shape check while never terminating. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Prepared = visual_xsharp::core; + + constexpr std::uint64_t kTotal = 2U; + constexpr std::uint64_t kIndex = 3U; + + [[nodiscard]] auto + Integer(std::int64_t value) -> Core::Expression + { + return Core::Expression::Constant(value, Core::Type::int64()); + } + + [[nodiscard]] auto + Spelling(std::uint64_t id) -> std::u32string + { + return id == kTotal ? U"total" : id == kIndex ? U"index" : U"inner"; + } + + [[nodiscard]] auto + Variable(std::uint64_t id, std::u32string spelling) -> Core::Expression + { + return Core::Expression::Variable({ id, std::move(spelling) }, + Core::Type::int64()); + } + + [[nodiscard]] auto + Compare(Core::Primitive operation, std::uint64_t id, std::int64_t value) + -> Core::Expression + { + return Core::Expression::InvokePrimitive( + operation, + { Variable(id, Spelling(id)), Integer(value) }, + Core::Type::boolean()); + } + + [[nodiscard]] auto + Increment(std::uint64_t id, std::u32string spelling) -> Core::Statement + { + return Core::Statement::Assign( + { id, spelling }, + Core::Expression::InvokePrimitive( + Core::Primitive::Add, + { Variable(id, spelling), Integer(1) }, + Core::Type::int64())); + } + + [[nodiscard]] auto + Accumulate() -> Core::Statement + { + return Core::Statement::Assign( + { kTotal, U"total" }, + Core::Expression::InvokePrimitive( + Core::Primitive::Add, + { Variable(kTotal, U"total"), Variable(kIndex, U"index") }, + Core::Type::int64())); + } + + /// `int total = 0; int index = 0; ; return total;` + [[nodiscard]] auto + LoopModule(Core::Statement loop) -> Core::Module + { + Core::Function function{ + { 1U, U"Evaluate" }, + {}, + Core::Type::int64(), + { Core::Statement::Bind({ { kTotal, U"total" }, + Core::Type::int64(), + true, + Integer(0) }), + Core::Statement::Bind({ { kIndex, U"index" }, + Core::Type::int64(), + true, + Integer(0) }), + std::move(loop), + Core::Statement::Return(Variable(kTotal, U"total")) }, + }; + return { { U"Loops" }, { std::move(function) } }; + } + + [[nodiscard]] auto + PrepareVerified(const Core::Module &module) -> Prepared::Function + { + for (const auto &issue : Core::Verify(module)) + FAIL_CHECK("Core " << issue.code << ": " << issue.message); + REQUIRE(Core::Verify(module).empty()); + auto prepared = Core::CorePrep::Prepare(module); + for (const auto &issue : Prepared::verify(prepared)) + FAIL_CHECK("CorePrep " << issue.code << ": " << issue.message + << " (block " << issue.block << ")"); + REQUIRE(Prepared::verify(prepared).empty()); + REQUIRE(prepared.functions.size() == 1U); + return std::move(prepared.functions.front()); + } + + [[nodiscard]] auto + Find(const Prepared::Function &function, Prepared::BlockId id) + -> const Prepared::Block & + { + const auto found + = std::ranges::find(function.blocks, id, &Prepared::Block::id); + REQUIRE(found != function.blocks.end()); + return *found; + } + + [[nodiscard]] auto + Successors(const Prepared::Block &block) -> std::vector + { + switch (block.terminator.kind) + { + case Prepared::Terminator::Kind::Jump: + return { block.terminator.true_target }; + case Prepared::Terminator::Kind::Branch: + return { block.terminator.true_target, + block.terminator.false_target }; + default: + return {}; + } + } + + /// Blocks reachable from the entry that end in a function return. + [[nodiscard]] auto + ReachesReturn(const Prepared::Function &function, Prepared::BlockId from) + -> bool + { + std::vector pending{ from }; + std::vector seen; + while (!pending.empty()) + { + const auto id = pending.back(); + pending.pop_back(); + if (std::ranges::find(seen, id) != seen.end()) + continue; + seen.push_back(id); + const auto &block = Find(function, id); + if (block.terminator.kind == Prepared::Terminator::Kind::Return) + return true; + for (const auto successor : Successors(block)) + pending.push_back(successor); + } + return false; + } + + [[nodiscard]] auto + IsJumpTo(const Prepared::Block &block, Prepared::BlockId target) -> bool + { + return block.terminator.kind == Prepared::Terminator::Kind::Jump + && block.terminator.true_target == target; + } + + struct ForShape final + { + Prepared::BlockId condition; + Prepared::BlockId body; + Prepared::BlockId update; + Prepared::BlockId exit; + }; + + /// Recover the for-loop block roles from edges alone, then check each. + [[nodiscard]] auto + ForShapeOf(const Prepared::Function &function) -> ForShape + { + const auto &entry = Find(function, function.entry); + REQUIRE(entry.terminator.kind == Prepared::Terminator::Kind::Jump); + const auto conditionId = entry.terminator.true_target; + const auto &condition = Find(function, conditionId); + REQUIRE(condition.terminator.kind + == Prepared::Terminator::Kind::Branch); + const auto bodyId = condition.terminator.true_target; + const auto exitId = condition.terminator.false_target; + // The adapter numbers the update region directly after the body + // entry; the edge checks below confirm that role independently. + return { conditionId, bodyId, bodyId + 1U, exitId }; + } +} // namespace + +TEST_CASE("for-loop update region returns to the condition block", + "[coreprep][loop]") +{ + const auto function = PrepareVerified(LoopModule( + Core::Statement::For(Compare(Core::Primitive::LessThan, kIndex, 1), + { Accumulate() }, + { Increment(kIndex, U"index") }))); + const auto shape = ForShapeOf(function); + + const auto &body = Find(function, shape.body); + const auto &update = Find(function, shape.update); + CHECK(IsJumpTo(body, shape.update)); + // The update region's only successor is the condition. A jump back to + // the update block itself is an infinite loop that skips the condition. + CHECK(IsJumpTo(update, shape.condition)); + CHECK_FALSE(IsJumpTo(update, shape.update)); + CHECK(update.instructions.size() >= 1U); + CHECK(Find(function, shape.exit).terminator.kind + == Prepared::Terminator::Kind::Return); + for (const auto &block : function.blocks) + CHECK(ReachesReturn(function, block.id)); +} + +TEST_CASE("for-loop continue runs the update and break leaves the loop", + "[coreprep][loop]") +{ + const auto function = PrepareVerified(LoopModule(Core::Statement::For( + Compare(Core::Primitive::LessThan, kIndex, 12), + { Core::Statement::If(Compare(Core::Primitive::Equal, kIndex, 2), + { Core::Statement::Continue() }, + {}), + Core::Statement::If(Compare(Core::Primitive::Equal, kIndex, 9), + { Core::Statement::Break() }, + {}), + Accumulate() }, + { Increment(kIndex, U"index") }))); + const auto shape = ForShapeOf(function); + + // Body entry tests `index == 2`; its true arm is the continue. + const auto &body = Find(function, shape.body); + REQUIRE(body.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(IsJumpTo(Find(function, body.terminator.true_target), shape.update)); + + // The join of the first `if` tests `index == 9`; its true arm breaks. + const auto &firstElse = Find(function, body.terminator.false_target); + REQUIRE(firstElse.terminator.kind == Prepared::Terminator::Kind::Jump); + const auto &second = Find(function, firstElse.terminator.true_target); + REQUIRE(second.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(IsJumpTo(Find(function, second.terminator.true_target), shape.exit)); + + const auto &update = Find(function, shape.update); + CHECK(IsJumpTo(update, shape.condition)); + + // Exactly the body tail and the continue arm enter the update region, + // and only the update region re-enters the condition from inside. + std::size_t intoUpdate{}; + std::size_t intoCondition{}; + for (const auto &block : function.blocks) + { + const auto successors = Successors(block); + intoUpdate += static_cast( + std::ranges::count(successors, shape.update)); + intoCondition += static_cast( + std::ranges::count(successors, shape.condition)); + CHECK(ReachesReturn(function, block.id)); + } + CHECK(intoUpdate == 2U); + CHECK(intoCondition == 2U); // function entry and the update region +} + +TEST_CASE("for-loop with an empty update still returns to the condition", + "[coreprep][loop]") +{ + const auto function = PrepareVerified(LoopModule( + Core::Statement::For(Compare(Core::Primitive::LessThan, kIndex, 3), + { Accumulate(), Increment(kIndex, U"index") }, + {}))); + const auto shape = ForShapeOf(function); + const auto &update = Find(function, shape.update); + CHECK(update.instructions.empty()); + CHECK(IsJumpTo(update, shape.condition)); +} + +TEST_CASE("for-loop update containing a branch closes every open tail", + "[coreprep][loop]") +{ + // Structured updates may lower to several blocks; each open tail must + // reach the condition, not only the first update block. + const auto function = PrepareVerified(LoopModule(Core::Statement::For( + Compare(Core::Primitive::LessThan, kIndex, 4), + { Accumulate() }, + { Core::Statement::If( + Compare(Core::Primitive::LessThan, kTotal, 2), + { Increment(kIndex, U"index") }, + { Increment(kIndex, U"index"), Increment(kTotal, U"total") }) }))); + const auto shape = ForShapeOf(function); + const auto &update = Find(function, shape.update); + REQUIRE(update.terminator.kind == Prepared::Terminator::Kind::Branch); + const auto &join = Find( + function, + Find(function, update.terminator.true_target).terminator.true_target); + CHECK(IsJumpTo(join, shape.condition)); + for (const auto &block : function.blocks) + { + CHECK_FALSE(IsJumpTo(block, block.id)); + CHECK(ReachesReturn(function, block.id)); + } +} + +TEST_CASE("for-loop numeric condition keeps its comparison in the header", + "[coreprep][loop]") +{ + // A non-Boolean condition is canonicalized to `value != 0`. That + // comparison belongs to the condition block so every iteration + // re-evaluates it and the branch operand is defined on all paths. + const auto function = PrepareVerified(LoopModule( + Core::Statement::For(Core::Expression::InvokePrimitive( + Core::Primitive::Subtract, + { Integer(3), Variable(kIndex, U"index") }, + Core::Type::int64()), + { Accumulate() }, + { Increment(kIndex, U"index") }))); + const auto shape = ForShapeOf(function); + const auto &condition = Find(function, shape.condition); + REQUIRE(condition.terminator.value.kind == Prepared::Atom::Kind::Variable); + const auto branchSymbol = condition.terminator.value.symbol.id; + CHECK(condition.terminator.value.type == Core::Type::boolean()); + CHECK(std::ranges::any_of( + condition.instructions, + [branchSymbol](const auto &instruction) { + return instruction.destination.id == branchSymbol + && instruction.operation == Prepared::Operation::NotEqual; + })); +} + +TEST_CASE("nested for-loops keep independent continue and break targets", + "[coreprep][loop]") +{ + constexpr std::uint64_t kInner = 4U; + auto inner = Core::Statement::For( + Compare(Core::Primitive::LessThan, kInner, 3), + { Core::Statement::If(Compare(Core::Primitive::Equal, kInner, 1), + { Core::Statement::Continue() }, + {}), + Accumulate() }, + { Increment(kInner, U"inner") }); + const auto function = PrepareVerified(LoopModule(Core::Statement::For( + Compare(Core::Primitive::LessThan, kIndex, 3), + { Core::Statement::Bind( + { { kInner, U"inner" }, Core::Type::int64(), true, Integer(0) }), + std::move(inner) }, + { Increment(kIndex, U"index") }))); + const auto outer = ForShapeOf(function); + + const auto &outerBody = Find(function, outer.body); + REQUIRE(outerBody.terminator.kind == Prepared::Terminator::Kind::Jump); + const auto innerConditionId = outerBody.terminator.true_target; + const auto &innerCondition = Find(function, innerConditionId); + REQUIRE(innerCondition.terminator.kind + == Prepared::Terminator::Kind::Branch); + const auto innerBodyId = innerCondition.terminator.true_target; + const auto innerUpdateId = innerBodyId + 1U; + const auto innerExitId = innerCondition.terminator.false_target; + + CHECK(IsJumpTo(Find(function, innerUpdateId), innerConditionId)); + // The inner continue targets the inner update, never the outer one. + const auto &innerBody = Find(function, innerBodyId); + REQUIRE(innerBody.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(IsJumpTo(Find(function, innerBody.terminator.true_target), + innerUpdateId)); + // Leaving the inner loop falls through to the outer update region. + CHECK(IsJumpTo(Find(function, innerExitId), outer.update)); + CHECK(IsJumpTo(Find(function, outer.update), outer.condition)); + for (const auto &block : function.blocks) + CHECK(ReachesReturn(function, block.id)); +} + +TEST_CASE("while-loop condition owns a header separate from prior statements", + "[coreprep][loop]") +{ + const auto function = PrepareVerified(LoopModule(Core::Statement::While( + Compare(Core::Primitive::LessThan, kIndex, 3), + { Core::Statement::If( + Compare(Core::Primitive::Equal, kIndex, 1), + { Increment(kIndex, U"index"), Core::Statement::Continue() }, + {}), + Accumulate(), + Increment(kIndex, U"index") }))); + + const auto &entry = Find(function, function.entry); + REQUIRE(entry.terminator.kind == Prepared::Terminator::Kind::Jump); + const auto headerId = entry.terminator.true_target; + REQUIRE(headerId != function.entry); + const auto &header = Find(function, headerId); + REQUIRE(header.terminator.kind == Prepared::Terminator::Kind::Branch); + + // The initializers stay in the entry block and are never re-executed: + // nothing in the header may define or assign a source variable. + CHECK(entry.instructions.size() == 2U); + CHECK( + std::ranges::none_of(header.instructions, [](const auto &instruction) { + return instruction.destination.id == kTotal + || instruction.destination.id == kIndex; + })); + + std::size_t backEdges{}; + for (const auto &block : function.blocks) + { + if (block.id != function.entry) + backEdges += static_cast( + std::ranges::count(Successors(block), headerId)); + CHECK(std::ranges::count(Successors(block), function.entry) == 0); + CHECK(ReachesReturn(function, block.id)); + } + CHECK(backEdges == 2U); // the continue arm and the body tail + + const auto &body = Find(function, header.terminator.true_target); + REQUIRE(body.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(IsJumpTo(Find(function, body.terminator.true_target), headerId)); +} + +TEST_CASE("do-while continue targets the trailing condition block", + "[coreprep][loop]") +{ + const auto function = PrepareVerified(LoopModule(Core::Statement::DoWhile( + { Increment(kIndex, U"index"), + Core::Statement::If(Compare(Core::Primitive::Equal, kIndex, 2), + { Core::Statement::Continue() }, + {}), + Core::Statement::If(Compare(Core::Primitive::Equal, kIndex, 5), + { Core::Statement::Break() }, + {}), + Accumulate() }, + Compare(Core::Primitive::LessThan, kIndex, 8)))); + + const auto &entry = Find(function, function.entry); + REQUIRE(entry.terminator.kind == Prepared::Terminator::Kind::Jump); + const auto bodyId = entry.terminator.true_target; + const auto conditionId = bodyId + 1U; + const auto &condition = Find(function, conditionId); + REQUIRE(condition.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(condition.terminator.true_target == bodyId); + const auto exitId = condition.terminator.false_target; + + const auto &body = Find(function, bodyId); + REQUIRE(body.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(IsJumpTo(Find(function, body.terminator.true_target), conditionId)); + const auto &second = Find( + function, + Find(function, body.terminator.false_target).terminator.true_target); + REQUIRE(second.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(IsJumpTo(Find(function, second.terminator.true_target), exitId)); + for (const auto &block : function.blocks) + CHECK(ReachesReturn(function, block.id)); +} diff --git a/Compiler/Core/Tests/ShortCircuitLoweringTests.cpp b/Compiler/Core/Tests/ShortCircuitLoweringTests.cpp new file mode 100644 index 00000000..1e83ec96 --- /dev/null +++ b/Compiler/Core/Tests/ShortCircuitLoweringTests.cpp @@ -0,0 +1,327 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Core/CorePrep/Prepare.hpp" +#include "Visual/XSharp/Core/CorePrep/Verifier.hpp" +#include "Visual/XSharp/Core/Verifier.hpp" + +// `&&` and `||` are control flow in CorePrep: the right operand is reached +// only through the branch that needs it. These tests pin that shape on the +// native Core-to-CorePrep adapter. An eager two-operand instruction computes +// the same Boolean for pure operands but evaluates a guarded division, call +// or recursion that the source never executes. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Prepared = visual_xsharp::core; + + constexpr std::uint64_t kLeft = 2U; + constexpr std::uint64_t kRight = 3U; + constexpr std::uint64_t kThird = 4U; + + [[nodiscard]] auto + Spelling(std::uint64_t id) -> std::u32string + { + return id == kLeft ? U"left" : id == kRight ? U"right" : U"third"; + } + + [[nodiscard]] auto + Integer(std::int64_t value) -> Core::Expression + { + return Core::Expression::Constant(value, Core::Type::int64()); + } + + [[nodiscard]] auto + Variable(std::uint64_t id) -> Core::Expression + { + return Core::Expression::Variable({ id, Spelling(id) }, + Core::Type::int64()); + } + + /// `id < 10`; the distinct symbol identifies which operand a block owns. + [[nodiscard]] auto + Below(std::uint64_t id) -> Core::Expression + { + return Core::Expression::InvokePrimitive(Core::Primitive::LessThan, + { Variable(id), Integer(10) }, + Core::Type::boolean()); + } + + [[nodiscard]] auto + Logical(Core::Primitive operation, + Core::Expression left, + Core::Expression right) -> Core::Expression + { + return Core::Expression::InvokePrimitive( + operation, + { std::move(left), std::move(right) }, + Core::Type::boolean()); + } + + [[nodiscard]] auto + Declare(std::uint64_t id) -> Core::Statement + { + return Core::Statement::Bind( + { { id, Spelling(id) }, Core::Type::int64(), true, Integer(0) }); + } + + [[nodiscard]] auto + Module(Core::Type returnType, std::vector tail) + -> Core::Module + { + std::vector body{ Declare(kLeft), + Declare(kRight), + Declare(kThird) }; + body.insert(body.end(), + std::make_move_iterator(tail.begin()), + std::make_move_iterator(tail.end())); + return { { U"ShortCircuit" }, + { Core::Function{ { 1U, U"Evaluate" }, + {}, + std::move(returnType), + std::move(body) } } }; + } + + [[nodiscard]] auto + PrepareVerified(const Core::Module &module) -> Prepared::Function + { + for (const auto &issue : Core::Verify(module)) + FAIL_CHECK("Core " << issue.code << ": " << issue.message); + REQUIRE(Core::Verify(module).empty()); + auto prepared = Core::CorePrep::Prepare(module); + for (const auto &issue : Prepared::verify(prepared)) + FAIL_CHECK("CorePrep " << issue.code << ": " << issue.message + << " (block " << issue.block << ")"); + REQUIRE(Prepared::verify(prepared).empty()); + REQUIRE(prepared.functions.size() == 1U); + return std::move(prepared.functions.front()); + } + + [[nodiscard]] auto + Find(const Prepared::Function &function, Prepared::BlockId id) + -> const Prepared::Block & + { + const auto found + = std::ranges::find(function.blocks, id, &Prepared::Block::id); + REQUIRE(found != function.blocks.end()); + return *found; + } + + /// The block whose instructions read the given source variable. + [[nodiscard]] auto + BlockReading(const Prepared::Function &function, std::uint64_t symbol) + -> std::optional + { + for (const auto &block : function.blocks) + for (const auto &instruction : block.instructions) + if (instruction.operation == Prepared::Operation::LessThan + && std::ranges::any_of( + instruction.operands, + [symbol](const auto &operand) { + return operand.kind + == Prepared::Atom::Kind::Variable + && operand.symbol.id == symbol; + })) + return block.id; + return std::nullopt; + } + + [[nodiscard]] auto + HasEagerLogicalOperation(const Prepared::Function &function) -> bool + { + return std::ranges::any_of(function.blocks, [](const auto &block) { + return std::ranges::any_of( + block.instructions, + [](const auto &instruction) { + return instruction.operation + == Prepared::Operation::LogicalAnd + || instruction.operation + == Prepared::Operation::LogicalOr; + }); + }); + } + + [[nodiscard]] auto + IsJumpTo(const Prepared::Block &block, Prepared::BlockId target) -> bool + { + return block.terminator.kind == Prepared::Terminator::Kind::Jump + && block.terminator.true_target == target; + } +} // namespace + +TEST_CASE("logical and evaluates its right operand only on the true edge", + "[coreprep][shortcircuit]") +{ + const auto function = PrepareVerified( + Module(Core::Type::boolean(), + { Core::Statement::Return(Logical(Core::Primitive::LogicalAnd, + Below(kLeft), + Below(kRight))) })); + CHECK_FALSE(HasEagerLogicalOperation(function)); + + const auto &entry = Find(function, function.entry); + REQUIRE(entry.terminator.kind == Prepared::Terminator::Kind::Branch); + const auto leftBlock = BlockReading(function, kLeft); + const auto rightBlock = BlockReading(function, kRight); + REQUIRE(leftBlock); + REQUIRE(rightBlock); + CHECK(*leftBlock == function.entry); + CHECK(*rightBlock != function.entry); + // True continues into the right operand; false skips it to the join. + CHECK(entry.terminator.true_target == *rightBlock); + const auto joinId = entry.terminator.false_target; + CHECK(joinId != *rightBlock); + const auto &right = Find(function, *rightBlock); + CHECK(IsJumpTo(right, joinId)); + + // The join returns one slot: initialized false before the branch and + // overwritten only by the right operand's block. + const auto &join = Find(function, joinId); + REQUIRE(join.terminator.kind == Prepared::Terminator::Kind::Return); + REQUIRE(join.terminator.value.kind == Prepared::Atom::Kind::Variable); + const auto result = join.terminator.value.symbol.id; + const auto initializer + = std::ranges::find_if(entry.instructions, + [result](const auto &instruction) { + return instruction.destination.id == result; + }); + REQUIRE(initializer != entry.instructions.end()); + CHECK(initializer->kind == Prepared::Instruction::Kind::Bind); + CHECK(initializer->mutable_binding); + REQUIRE(initializer->operands.size() == 1U); + CHECK(initializer->operands.front().literal == Prepared::Literal{ false }); + REQUIRE_FALSE(right.instructions.empty()); + CHECK(right.instructions.back().kind + == Prepared::Instruction::Kind::Assign); + CHECK(right.instructions.back().destination.id == result); +} + +TEST_CASE("logical or evaluates its right operand only on the false edge", + "[coreprep][shortcircuit]") +{ + const auto function = PrepareVerified( + Module(Core::Type::boolean(), + { Core::Statement::Return(Logical(Core::Primitive::LogicalOr, + Below(kLeft), + Below(kRight))) })); + CHECK_FALSE(HasEagerLogicalOperation(function)); + + const auto &entry = Find(function, function.entry); + REQUIRE(entry.terminator.kind == Prepared::Terminator::Kind::Branch); + const auto rightBlock = BlockReading(function, kRight); + REQUIRE(rightBlock); + CHECK(entry.terminator.false_target == *rightBlock); + const auto joinId = entry.terminator.true_target; + CHECK(IsJumpTo(Find(function, *rightBlock), joinId)); + const auto &join = Find(function, joinId); + REQUIRE(join.terminator.kind == Prepared::Terminator::Kind::Return); + const auto result = join.terminator.value.symbol.id; + const auto initializer + = std::ranges::find_if(entry.instructions, + [result](const auto &instruction) { + return instruction.destination.id == result; + }); + REQUIRE(initializer != entry.instructions.end()); + REQUIRE(initializer->operands.size() == 1U); + CHECK(initializer->operands.front().literal == Prepared::Literal{ true }); +} + +TEST_CASE("nested short-circuit operands stay behind their own guards", + "[coreprep][shortcircuit]") +{ + // (left && right) || third: `right` needs left, `third` needs the + // conjunction to be false, and neither is evaluated in the entry block. + const auto function = PrepareVerified(Module( + Core::Type::boolean(), + { Core::Statement::Return(Logical( + Core::Primitive::LogicalOr, + Logical(Core::Primitive::LogicalAnd, Below(kLeft), Below(kRight)), + Below(kThird))) })); + CHECK_FALSE(HasEagerLogicalOperation(function)); + const auto leftBlock = BlockReading(function, kLeft); + const auto rightBlock = BlockReading(function, kRight); + const auto thirdBlock = BlockReading(function, kThird); + REQUIRE(leftBlock); + REQUIRE(rightBlock); + REQUIRE(thirdBlock); + CHECK(*leftBlock == function.entry); + CHECK(*rightBlock != function.entry); + CHECK(*thirdBlock != function.entry); + CHECK(*thirdBlock != *rightBlock); + + const auto &entry = Find(function, function.entry); + REQUIRE(entry.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(entry.terminator.true_target == *rightBlock); + // The conjunction's join decides whether `third` runs at all. + const auto &innerJoin = Find(function, entry.terminator.false_target); + REQUIRE(innerJoin.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(innerJoin.terminator.false_target == *thirdBlock); + CHECK(IsJumpTo(Find(function, *thirdBlock), + innerJoin.terminator.true_target)); +} + +TEST_CASE("short-circuit loop condition is re-evaluated from the loop header", + "[coreprep][shortcircuit][loop]") +{ + const auto function = PrepareVerified(Module( + Core::Type::int64(), + { Core::Statement::While( + Logical(Core::Primitive::LogicalAnd, Below(kLeft), Below(kRight)), + { Core::Statement::Assign({ kLeft, Spelling(kLeft) }, + Core::Expression::InvokePrimitive( + Core::Primitive::Add, + { Variable(kLeft), Integer(1) }, + Core::Type::int64())) }), + Core::Statement::Return(Variable(kLeft)) })); + CHECK_FALSE(HasEagerLogicalOperation(function)); + + const auto &entry = Find(function, function.entry); + REQUIRE(entry.terminator.kind == Prepared::Terminator::Kind::Jump); + const auto headerId = entry.terminator.true_target; + const auto leftBlock = BlockReading(function, kLeft); + const auto rightBlock = BlockReading(function, kRight); + REQUIRE(leftBlock); + REQUIRE(rightBlock); + CHECK(*leftBlock == headerId); + CHECK(*rightBlock != headerId); + + // The header only decides whether the right operand runs; the loop's + // own body/exit branch lives in the short-circuit join. + const auto &header = Find(function, headerId); + REQUIRE(header.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(header.terminator.true_target == *rightBlock); + const auto &join = Find(function, header.terminator.false_target); + REQUIRE(join.terminator.kind == Prepared::Terminator::Kind::Branch); + const auto &body = Find(function, join.terminator.true_target); + // The back-edge re-enters the header so `left` is tested again. + CHECK(IsJumpTo(body, headerId)); + CHECK(Find(function, join.terminator.false_target).terminator.kind + == Prepared::Terminator::Kind::Return); +} + +TEST_CASE("logical not remains an ordinary single-block operation", + "[coreprep][shortcircuit]") +{ + const auto function = PrepareVerified( + Module(Core::Type::boolean(), + { Core::Statement::Return(Core::Expression::InvokePrimitive( + Core::Primitive::LogicalNot, + { Below(kLeft) }, + Core::Type::boolean())) })); + REQUIRE(function.blocks.size() == 1U); + CHECK(std::ranges::any_of(function.blocks.front().instructions, + [](const auto &instruction) { + return instruction.operation + == Prepared::Operation::LogicalNot; + })); +} diff --git a/Compiler/Fuzzing/Corpus/cli/options.seed b/Compiler/Fuzzing/Corpus/cli/options.seed new file mode 100644 index 00000000..c3081393 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/cli/options.seed @@ -0,0 +1,3 @@ +build +-Help +-- diff --git a/Compiler/Fuzzing/Corpus/ownership/race.seed b/Compiler/Fuzzing/Corpus/ownership/race.seed new file mode 100644 index 00000000..30f4b487 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/ownership/race.seed @@ -0,0 +1 @@ +strong-weak-unowned-race diff --git a/Compiler/Fuzzing/Corpus/project/registry.seed b/Compiler/Fuzzing/Corpus/project/registry.seed new file mode 100644 index 00000000..72bcbb43 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/project/registry.seed @@ -0,0 +1 @@ +visual-xsharp-sources-v6 diff --git a/Compiler/Fuzzing/Corpus/repl/history.seed b/Compiler/Fuzzing/Corpus/repl/history.seed new file mode 100644 index 00000000..07ea8213 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/repl/history.seed @@ -0,0 +1 @@ +repl-session-reset-type-error-value diff --git a/Compiler/Fuzzing/SourceFuzz.cpp b/Compiler/Fuzzing/SourceFuzz.cpp index 559411bc..009250b8 100644 --- a/Compiler/Fuzzing/SourceFuzz.cpp +++ b/Compiler/Fuzzing/SourceFuzz.cpp @@ -2,8 +2,10 @@ // SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 #include +#include #include #include +#include #include #include #include @@ -78,12 +80,89 @@ namespace Visual::XSharp::Fuzzing const auto expression = GenerateExpression(bytes, cursor, kMaximumGeneratedDepth); expected = expression.value; + std::string body = "return " + expression.source + ";"; + std::string members; + const auto mode = NextByte(bytes, cursor) % 6U; + const auto limit + = static_cast(NextByte(bytes, cursor) % 12U); + if (mode == 1U) + { + // Independent host execution models for/continue/break rather + // than comparing two copies of the compiler's CFG algorithm. + expected = 0; + for (std::int64_t index = 0; index < limit; ++index) + { + if (index == 2) + continue; + if (index == 9) + break; + expected += index; + } + body = "int total = 0; for (int index = 0; index < " + + std::to_string(limit) + + "; index++) { if (index == 2) { continue; } " + "if (index == 9) { break; } total = total + index; } " + "return total;"; + } + else if (mode == 2U) + { + expected = limit == 0 ? 1 : limit; + body = "int total = 0; do { total++; } while (total < " + + std::to_string(limit) + "); return total;"; + } + else if (mode == 3U) + { + expected = limit < 6 ? expression.value : -expression.value; + body = "if (" + std::to_string(limit) + " < 6) { return " + + expression.source + "; } else { return -(" + + expression.source + "); }"; + } + else if (mode == 4U) + { + // Statements before a pre-test loop run exactly once. The + // initializers and the loop share one source block, so a + // back-edge that re-enters that block resets the counter. + expected = 0; + std::int64_t index = 0; + while (index < limit) + { + if (index == 1) + { + ++index; + continue; + } + if (index == 7) + break; + expected += index; + ++index; + } + body = "int total = 0; int index = 0; while (index < " + + std::to_string(limit) + + ") { if (index == 1) { index++; continue; } " + "if (index == 7) { break; } total = total + index; " + "index++; } return total;"; + } + else if (mode == 5U) + { + // The recursion ends only because `||` skips its right + // operand at zero, and the comparison after `&&` is reached + // only when the call returned. Evaluating either right + // operand eagerly recurses without bound. + expected = limit < 6 ? limit : -1; + members + = " public static bool Down(_ int n) { return n == 0 " + "|| Down(n - 1); }\n"; + body = "if (Down(" + std::to_string(limit) + ") && " + + std::to_string(limit) + " < 6) { return " + + std::to_string(limit) + "; } return 0 - 1;"; + } return "namespace Fuzz;\n" "class Program {\n" - " public static int Evaluate() {\n" - " return " - + expression.source - + ";\n" + + members + + " public static int Evaluate() {\n" + " " + + body + + "\n" " }\n" "}\n"; } @@ -181,6 +260,9 @@ namespace Visual::XSharp::Fuzzing Visual::XSharp::Pipeline::Options options; options.optimize_xpp = optimizeXpp; options.optimize_xmm = optimizeXmm; + options.llvm.optimization = optimizeXmm + ? Llvm::OptimizationLevel::Default + : Llvm::OptimizationLevel::Debug; const auto pipeline = Visual::XSharp::Pipeline::ConsumeCore(coreBytes, options); if (!pipeline || !pipeline.llvm) @@ -291,6 +373,10 @@ namespace Visual::XSharp::Fuzzing // independent evidence and repeats work in the expensive oracle. const auto unoptimized = CompileVariant(compiled.bytes, false, false); const auto optimized = CompileVariant(compiled.bytes, true, true); + if (std::getenv("VXS_FUZZ_TRACE") != nullptr) + llvm::errs() << source << "\nReference LLVM:\n" + << unoptimized.llvm_ir << "\nOptimized LLVM:\n" + << optimized.llvm_ir; constexpr std::string_view kReferenceModule = "vxs-fuzz-reference"; constexpr std::string_view kOptimizedModule = "vxs-fuzz-optimized"; const auto referenceValue = Invoke(unoptimized, kReferenceModule); diff --git a/Compiler/Fuzzing/SourceFuzzSmoke.cpp b/Compiler/Fuzzing/SourceFuzzSmoke.cpp index a28e48b6..52ae85bb 100644 --- a/Compiler/Fuzzing/SourceFuzzSmoke.cpp +++ b/Compiler/Fuzzing/SourceFuzzSmoke.cpp @@ -3,6 +3,7 @@ #include #include +#include #include #include "SourceFuzz.hpp" @@ -23,7 +24,9 @@ main() const std::span emptySource; Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(emptySource); Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(source); + llvm::errs() << "Differential smoke: mixed seed\n"; Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(expressionSeed); + llvm::errs() << "Differential smoke: empty seed\n"; Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(emptySource); // Constant seeds force leaves, full-depth addition, subtraction and // multiplication. Exercise the independent oracle before a mutation @@ -32,7 +35,26 @@ main() for (const auto selector : selectors) { const std::array seed{ selector }; + llvm::errs() << "Differential smoke: selector " + << static_cast(selector) << '\n'; Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(seed); } + // A leaf selector followed by an explicit mode and limit byte reaches + // every generated control-flow shape at every trip count, including the + // zero-trip, continue and break paths, instead of only the shapes the + // four cycling selectors above happen to select. + constexpr std::uint8_t kModes = 6U; + constexpr std::uint8_t kLimits = 12U; + for (std::uint8_t mode = 0U; mode < kModes; ++mode) + { + for (std::uint8_t limit = 0U; limit < kLimits; ++limit) + { + const std::array seed{ 0U, mode, limit }; + llvm::errs() << "Differential smoke: mode " + << static_cast(mode) << " limit " + << static_cast(limit) << '\n'; + Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(seed); + } + } return 0; } diff --git a/Compiler/Haskell/Driver/Fuzzing/Feedback.hs b/Compiler/Haskell/Driver/Fuzzing/Feedback.hs new file mode 100644 index 00000000..be3edc02 --- /dev/null +++ b/Compiler/Haskell/Driver/Fuzzing/Feedback.hs @@ -0,0 +1,39 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Exact HPC tick identities for frontend mutation feedback. Module hashes +keep different builds' counters distinct; each execution resets counters before +running so coverage is attributed to that input rather than accumulated history. +-} +module Feedback (Coverage, coverageFor, availableTicks, requiredModulePresent) where + +import Data.List (isInfixOf) +import Data.Set qualified as Set +import Trace.Hpc.Tix (Tix (..), TixModule (..)) + +type Coverage = Set.Set (String, String, Int) + +ownedModule :: String -> Bool +ownedModule = isInfixOf "Visual.XSharp." + +coverageFor :: Tix -> Coverage +coverageFor (Tix modules) = + Set.fromList + [ (name, show identity, index) + | TixModule name identity _ ticks <- modules + , ownedModule name + , (index, count) <- zip [0 ..] ticks + , count > 0 + ] + +availableTicks :: Tix -> Int +availableTicks (Tix modules) = sum [size | TixModule name _ size _ <- modules, ownedModule name] + +requiredModulePresent :: String -> Tix -> Bool +requiredModulePresent stage (Tix modules) = + any (\(TixModule name _ size _) -> size > 0 && isInfixOf expected name) modules + where + expected = case stage of + "lexer" -> "Visual.XSharp.Lexer" + "parser" -> "Visual.XSharp.Parser" + _ -> "Visual.XSharp.Compiler" diff --git a/Compiler/Haskell/Driver/Fuzzing/Main.hs b/Compiler/Haskell/Driver/Fuzzing/Main.hs new file mode 100644 index 00000000..04179fde --- /dev/null +++ b/Compiler/Haskell/Driver/Fuzzing/Main.hs @@ -0,0 +1,182 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +{-# LANGUAGE BangPatterns #-} + +module Main (main) where + +import Control.Exception (SomeException, displayException, evaluate, try) +import Control.Monad (forM, unless, when) +import Data.ByteString qualified as BS +import Data.List (sort) +import Data.Sequence qualified as Seq +import Data.Set qualified as Set +import Data.Text qualified as Text +import Data.Text.Encoding qualified as Text +import Data.Word (Word64) +import Feedback +import GHC.Clock (getMonotonicTimeNSec) +import Mutation +import System.Directory + ( createDirectoryIfMissing + , doesFileExist + , getFileSize + , listDirectory + , pathIsSymbolicLink + , renameFile + ) +import System.Environment (getArgs) +import System.Exit (die) +import System.FilePath (()) +import System.IO (hClose, openBinaryTempFile) +import System.IO.Error (catchIOError, isDoesNotExistError) +import System.Timeout (timeout) +import Text.Read (readMaybe) +import Trace.Hpc.Reflect (clearTix, examineTix) +import Visual.XSharp.Compiler +import Visual.XSharp.Frontend (analyzeSyntax) +import Visual.XSharp.Lexer + +main :: IO () +main = do + arguments <- getArgs + case arguments of + [stage, "--replay", path] -> do + validateStage stage + input <- BS.readFile path + when (BS.length input > maximumInput) (die "replay input exceeds the campaign limit") + execute stage input >>= either die (const (putStrLn "HPC replay completed")) + [stage, secondsText, corpus, artifacts, seedText] + | Just seconds <- readMaybe secondsText + , seconds >= (1 :: Int) && seconds <= 3600 + , Just seed <- readMaybe seedText -> do + validateStage stage + campaign stage seconds corpus artifacts seed + _ -> die "usage: frontend-fuzz STAGE SECONDS CORPUS ARTIFACTS SEED | STAGE --replay FILE" + +validateStage :: String -> IO () +validateStage stage = unless (stage `elem` ["lexer", "parser", "source"]) (die "unknown HPC fuzz stage") + +execute :: String -> BS.ByteString -> IO (Either String Coverage) +execute stage input = do + -- Timeout surrounds exception capture so ordinary lazy failures cannot + -- escape, and a timed-out input cannot be mistaken for a normal diagnostic. + clearTix + attempted <- try (timeout 5000000 action) :: IO (Either SomeException (Maybe Int)) + case attempted of + Left issue -> pure (Left (displayException issue)) + Right Nothing -> pure (Left "frontend input exceeded five seconds") + Right (Just _) -> Right . coverageFor <$> examineTix + where + action = case Text.decodeUtf8' input of + Left _ -> pure 0 + Right text -> do + let source = Text.unpack text + compilerInput = CompilerInput "Fuzz.vxs" source + evaluate $ length $ case stage of + "lexer" -> show (runLexer defaultLexer (LexerInput "Fuzz.vxs" source)) + "parser" -> show (analyzeSyntax compilerInput) + _ -> show (compileToCorePrep compilerInput) + +loadSeeds :: FilePath -> IO [BS.ByteString] +loadSeeds corpus = do + names <- sort <$> listDirectory corpus + inputs <- forM (take 4096 names) $ \name -> do + let path = corpus name + regular <- doesFileExist path + symbolic <- pathIsSymbolicLink path + if regular && not symbolic + then do + size <- getFileSize path + if size <= fromIntegral maximumInput then Just <$> BS.readFile path else pure Nothing + else pure Nothing + pure (BS.empty : [bytes | Just bytes <- inputs]) + +saveInput :: FilePath -> BS.ByteString -> IO () +saveInput corpus input = do + let path = corpus ("hpc-" ++ fingerprint input ++ ".seed") + symbolic <- + pathIsSymbolicLink path `catchIOError` \issue -> + if isDoesNotExistError issue then pure False else ioError issue + when symbolic (die "cached HPC corpus entry is a symbolic link") + exists <- doesFileExist path + if exists + then do + previous <- BS.readFile path + unless (previous == input) (die "HPC corpus fingerprint collision; refusing to replace the existing input") + else do + -- Rename a newly created regular file instead of opening the final + -- cached name for writing. A dangling/replaced symlink must never + -- redirect writes outside the single-writer campaign corpus. + (temporary, handle) <- openBinaryTempFile corpus ".pending-seed" + BS.hPut handle input + hClose handle + renameFile temporary path + +campaign :: String -> Int -> FilePath -> FilePath -> Word64 -> IO () +campaign stage seconds corpus artifacts initialState = do + createDirectoryIfMissing True corpus + createDirectoryIfMissing True artifacts + seeds <- loadSeeds corpus + -- Prime registration with an actually valid source before checking HPC. + -- A coverage-disabled build must fail rather than run unguided mutations. + let validSource = + BS.pack (map (fromIntegral . fromEnum) "namespace Fuzz; class Program { public static int Evaluate() { return 1; } }") + execute stage validSource >>= either die (const (pure ())) + initialTix <- examineTix + unless (requiredModulePresent stage initialTix) (die "frontend-fuzz requires HPC-instrumented production modules") + let tickCount = availableTicks initialTix + started <- getMonotonicTimeNSec + let deadline = started + fromIntegral seconds * 1000000000 + report executed coverage additions = + unlines + [ "stage=" ++ stage + , "executed_units=" ++ show executed + , "covered_ticks=" ++ show (Set.size coverage) + , "available_ticks=" ++ show tickCount + , "new_units_added=" ++ show additions + , "seed=" ++ show initialState + ] + finish executed coverage additions = do + unless (executed > 0 && not (Set.null coverage)) (die "HPC campaign did not execute instrumented production code") + let summary = report executed coverage additions + writeFile (artifacts "campaign.txt") summary + putStrLn ("HPC_FUZZ_RESULT\n" ++ summary) + failInput input problem = do + BS.writeFile (artifacts "failure.seed") input + writeFile (artifacts "failure.txt") (problem ++ "\nReplay: frontend-fuzz " ++ stage ++ " --replay failure.seed\n") + die ("HPC fuzz failure: " ++ problem) + addInput input pool coverage additions = do + checked <- execute stage input + case checked of + Left problem -> failInput input problem + Right observed -> do + let novel = not (observed `Set.isSubsetOf` coverage) + when novel (saveInput corpus input) + let nextPool = if novel && Seq.length pool < 4096 then pool Seq.|> input else pool + pure (nextPool, Set.union observed coverage, additions + if novel then 1 else 0) + warm pool coverage additions [] = pure (pool, coverage, additions) + warm pool coverage additions (input : remaining) = do + (nextPool, nextCoverage, nextAdditions) <- addInput input pool coverage additions + warm nextPool nextCoverage nextAdditions remaining + loop !state !pool !coverage !executed !additions = do + now <- getMonotonicTimeNSec + if now >= deadline + then finish executed coverage additions + else do + let next = nextState state + base = Seq.index pool (fromIntegral (state `mod` fromIntegral (Seq.length pool))) + partner = Seq.index pool (fromIntegral (next `mod` fromIntegral (Seq.length pool))) + input = mutate next base partner + (nextPool, nextCoverage, nextAdditions) <- addInput input pool coverage additions + -- Progress checkpoints retain mutation state for resource + -- termination; ordinary failures always save exact bytes. + when + (executed `mod` 256 == 0) + ( writeFile + (artifacts "progress.txt") + (report executed nextCoverage nextAdditions ++ "mutation_state=" ++ show next ++ "\n") + ) + loop next nextPool nextCoverage (executed + 1) nextAdditions + let warmSeeds = validSource : take 4096 seeds + (pool, coverage, additions) <- warm (Seq.singleton BS.empty) Set.empty (0 :: Int) warmSeeds + loop initialState pool coverage (length warmSeeds) additions diff --git a/Compiler/Haskell/Driver/Fuzzing/Mutation.hs b/Compiler/Haskell/Driver/Fuzzing/Mutation.hs new file mode 100644 index 00000000..0c2d2e77 --- /dev/null +++ b/Compiler/Haskell/Driver/Fuzzing/Mutation.hs @@ -0,0 +1,76 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Bounded deterministic byte mutations. A campaign records its initial state +and saves exact failing bytes, so both mutation sequences and individual failures +are reproducible. Word64 wrapping is intentional for this generator. +-} +module Mutation (mutate, nextState, fingerprint, maximumInput) where + +import Data.Bits (xor) +import Data.ByteString qualified as BS +import Data.ByteString.Char8 qualified as Char8 +import Data.Word (Word64) +import Numeric (showHex) + +maximumInput :: Int +maximumInput = 8192 + +nextState :: Word64 -> Word64 +nextState state = state * 6364136223846793005 + 1442695040888963407 + +fingerprint :: BS.ByteString -> String +fingerprint bytes = + showHex + (BS.foldl' (\hash byte -> (hash `xor` fromIntegral byte) * 1099511628211) (14695981039346656037 :: Word64) bytes) + "" + +dictionary :: [BS.ByteString] +dictionary = + map + Char8.pack + [ "namespace Fuzz;" + , "class Program {" + , "public static int Evaluate() {" + , "return " + , "int value = " + , "while (" + , "if (" + , "for (int i = 0; i < 3; i++) {" + , "break;" + , "continue;" + , "not " + , "\\=" + , "&&" + , "||" + , "**" + , "-- comment\n" + , "0" + , "1" + , "255" + , "65536" + , ";" + , "}" + , "(" + , ")" + , "\"" + , "\0" + , "\r\n" + ] + +mutate :: Word64 -> BS.ByteString -> BS.ByteString -> BS.ByteString +mutate state input partner = BS.take maximumInput result + where + size = BS.length input + position = fromIntegral (nextState state `mod` fromIntegral (size + 1)) + prefix = BS.take position input + suffix = BS.drop position input + byte = fromIntegral (nextState (nextState state) `mod` 256) + token = dictionary !! fromIntegral (nextState state `mod` fromIntegral (length dictionary)) + result = case state `mod` 6 of + 0 -> prefix <> BS.singleton byte <> BS.drop 1 suffix + 1 -> prefix <> token <> suffix + 2 -> prefix <> BS.drop (1 + fromIntegral (state `mod` 16)) suffix + 3 -> prefix <> BS.take 128 partner <> suffix + 4 -> prefix <> BS.take 64 suffix <> suffix + _ -> prefix <> BS.singleton byte <> suffix diff --git a/Compiler/Haskell/Driver/Fuzzing/Tests.hs b/Compiler/Haskell/Driver/Fuzzing/Tests.hs new file mode 100644 index 00000000..beb1b603 --- /dev/null +++ b/Compiler/Haskell/Driver/Fuzzing/Tests.hs @@ -0,0 +1,39 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +module Main (main) where + +import Control.Monad (unless) +import Data.ByteString qualified as BS +import Data.Set qualified as Set +import Feedback +import Mutation +import System.Exit (die) +import Trace.Hpc.Tix (Tix (..), TixModule (..)) +import Trace.Hpc.Util (toHash) + +main :: IO () +main = do + let ticks = Tix [TixModule "Visual.XSharp.Parser" (toHash (1 :: Int)) 3 [1, 0, 2], TixModule "Main" (toHash (2 :: Int)) 1 [1]] + changed = Tix [TixModule "Visual.XSharp.Parser" (toHash (3 :: Int)) 3 [1, 0, 2]] + unless + (Set.size (coverageFor ticks) == 2 && availableTicks ticks == 3) + (die "feedback included harness ticks or missed production ticks") + unless + (requiredModulePresent "parser" ticks && not (requiredModulePresent "lexer" ticks)) + (die "coverage-disabled stage passed its gate") + unless + (Set.null (Set.intersection (coverageFor ticks) (coverageFor changed))) + (die "different module hashes shared tick identities") + let large = BS.replicate maximumInput 255 + empty = BS.empty + unless + ( all + (\seed -> BS.length (mutate seed large large) <= maximumInput && BS.length (mutate seed empty empty) <= maximumInput) + [0 .. 10000] + ) + (die "mutation exceeded its input bound") + unless + (mutate 42 large empty == mutate 42 large empty && fingerprint empty /= fingerprint large) + (die "mutation/replay identity is inconsistent") + putStrLn "HPC feedback and mutation policy tests passed" diff --git a/Compiler/Haskell/Driver/visual-xsharp-compiler.cabal b/Compiler/Haskell/Driver/visual-xsharp-compiler.cabal index ba1444bb..47622271 100644 --- a/Compiler/Haskell/Driver/visual-xsharp-compiler.cabal +++ b/Compiler/Haskell/Driver/visual-xsharp-compiler.cabal @@ -93,6 +93,34 @@ foreign-library vxs-frontend visual-xsharp-compiler default-language: GHC2024 +executable frontend-fuzz + main-is: Main.hs + other-modules: Feedback Mutation + hs-source-dirs: Fuzzing + build-depends: + base >=4.20 && <4.23, + bytestring >=0.12 && <0.14, + containers >=0.7 && <0.9, + directory >=1.3 && <1.4, + filepath >=1.5 && <1.6, + hpc >=0.7 && <0.8, + text >=2.1 && <2.3, + visual-xsharp-compiler + ghc-options: -rtsopts "-with-rtsopts=-M512m -K8m" + default-language: GHC2024 + +test-suite fuzz-feedback-tests + type: exitcode-stdio-1.0 + main-is: Tests.hs + other-modules: Feedback Mutation + hs-source-dirs: Fuzzing + build-depends: + base >=4.20 && <4.23, + bytestring >=0.12 && <0.14, + containers >=0.7 && <0.9, + hpc >=0.7 && <0.8 + default-language: GHC2024 + test-suite visual-xsharp-compiler-tests type: exitcode-stdio-1.0 main-is: Main.hs diff --git a/Compiler/ProjectSystem/Bridge/Fuzzing/BUILD.bazel b/Compiler/ProjectSystem/Bridge/Fuzzing/BUILD.bazel new file mode 100644 index 00000000..96093643 --- /dev/null +++ b/Compiler/ProjectSystem/Bridge/Fuzzing/BUILD.bazel @@ -0,0 +1,7 @@ +load("@rules_cc//cc:cc_binary.bzl", "cc_binary") + +cc_binary( + name = "project_fuzzer", + srcs = ["ProjectFuzzer.cpp"], + deps = ["//Compiler/ProjectSystem/Bridge:project_driver", "@llvm//:llvm"], +) diff --git a/Compiler/ProjectSystem/Bridge/Fuzzing/ProjectFuzzer.cpp b/Compiler/ProjectSystem/Bridge/Fuzzing/ProjectFuzzer.cpp new file mode 100644 index 00000000..ef2a5cbf --- /dev/null +++ b/Compiler/ProjectSystem/Bridge/Fuzzing/ProjectFuzzer.cpp @@ -0,0 +1,81 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include + +#include "Compiler/ProjectSystem/Bridge/ProjectDriver.hpp" + +namespace +{ + void + Check(std::span bytes) + { + namespace Driver = Visual::XSharp::Driver; + const auto first = Driver::ParseProjectRegistry(bytes, false); + if (first != Driver::ParseProjectRegistry(bytes, false)) + llvm::report_fatal_error( + "project record decoding is nondeterministic"); + const auto required = Driver::ParseProjectRegistry(bytes, true); + if (required + && (!first + || (required->executables.empty() + && required->libraries.empty()))) + llvm::report_fatal_error( + "source-required project decoding lost its source contract"); + } +} // namespace + +extern "C" int +LLVMFuzzerTestOneInput(const std::uint8_t *data, std::size_t size) +{ + if (size > 65536U) + return 0; + Check(std::span(reinterpret_cast(data), size)); + // Reach past the version/header prefix on every input, including seeds + // that mutate one count to SIZE_MAX. No evaluator or filesystem runs here. + std::vector records{ "visual-xsharp-sources-v6", + "0.4.0", + "default", + "llvm", + "debug", + "all", + "true", + "false", + "false", + "false", + "true", + "true", + "true", + "0", + "aot", + "none", + "build/debug", + "0", + "1", + "0", + "0", + "Application", + "Example.Program", + "Sources", + "0" }; + if (size != 0U) + { + const auto selected + = static_cast(data[0]) % records.size(); + records[selected].assign(reinterpret_cast(data + 1U), + size - 1U); + } + std::string framed; + for (const auto &record : records) + { + framed.append(record); + framed.push_back('\0'); + } + Check(framed); + return 0; +} diff --git a/Compiler/ProjectSystem/Bridge/ProjectDriver.cpp b/Compiler/ProjectSystem/Bridge/ProjectDriver.cpp index 9b6c1c9f..3cf838d4 100644 --- a/Compiler/ProjectSystem/Bridge/ProjectDriver.cpp +++ b/Compiler/ProjectSystem/Bridge/ProjectDriver.cpp @@ -273,7 +273,7 @@ namespace Visual::XSharp::Driver if (!input) return std::nullopt; const auto end = static_cast(input.tellg()); - if (end <= 0 + if (end <= 0 || end > 4 * 1024 * 1024 || static_cast(end) > std::numeric_limits::max()) return std::nullopt; @@ -287,7 +287,7 @@ namespace Visual::XSharp::Driver } [[nodiscard]] std::optional> - SplitRecords(const std::vector &bytes) + SplitRecords(std::span bytes) { std::vector records; std::size_t start{}; @@ -330,7 +330,8 @@ namespace Visual::XSharp::Driver text->data() + text->size(), result); if (conversion.ec != std::errc{} - || conversion.ptr != text->data() + text->size()) + || conversion.ptr != text->data() + text->size() + || result > records_.size() - position_) return std::nullopt; return result; } @@ -441,7 +442,7 @@ namespace Visual::XSharp::Driver } [[nodiscard]] std::optional - ParseRegistry(const std::vector &bytes, bool requireSources) + ParseRegistry(std::span bytes, bool requireSources) { const auto split = SplitRecords(bytes); if (!split || split->size() < kHeaderRecordCount @@ -587,6 +588,15 @@ namespace Visual::XSharp::Driver } } // namespace + std::optional + ParseProjectRegistry(std::span bytes, bool requireSources) + { + if (bytes.empty() || bytes.size() > 4U * 1024U * 1024U + || bytes.back() != '\0') + return std::nullopt; + return ParseRegistry(bytes, requireSources); + } + std::optional ResolveProject(bool requireSources) { @@ -607,7 +617,7 @@ namespace Visual::XSharp::Driver "source registry\n"); return std::nullopt; } - auto project = ParseRegistry(*bytes, requireSources); + auto project = ParseProjectRegistry(*bytes, requireSources); if (!project) fmt::print(stderr, "vxs: bundled project evaluator returned invalid " diff --git a/Compiler/ProjectSystem/Bridge/ProjectDriver.hpp b/Compiler/ProjectSystem/Bridge/ProjectDriver.hpp index 62d5311e..02b76a46 100644 --- a/Compiler/ProjectSystem/Bridge/ProjectDriver.hpp +++ b/Compiler/ProjectSystem/Bridge/ProjectDriver.hpp @@ -5,6 +5,7 @@ #include #include +#include #include #include @@ -18,6 +19,8 @@ namespace Visual::XSharp::Driver std::optional framework; std::filesystem::path root; std::vector excludes; + bool + operator==(const ResolvedTestSuite &) const = default; }; struct ResolvedSourceTarget final @@ -28,6 +31,8 @@ namespace Visual::XSharp::Driver std::filesystem::path root; std::vector excludes; std::vector viPkgTypes; + bool + operator==(const ResolvedSourceTarget &) const = default; }; struct ResolvedProject final @@ -44,10 +49,16 @@ namespace Visual::XSharp::Driver std::filesystem::path outputDirectory; BuildOutput output{}; CompilerSettings settings{}; + bool + operator==(const ResolvedProject &) const = default; }; [[nodiscard]] std::optional ResolveProject(bool requireSources); + // Decode borrowed evaluator records without spawning a process. Every + // accepted field is copied into the returned project before bytes expire. + [[nodiscard]] std::optional + ParseProjectRegistry(std::span bytes, bool requireSources); [[nodiscard]] bool RefreshProjectLock(); } // namespace Visual::XSharp::Driver diff --git a/Compiler/ProjectSystem/Bridge/Tests/BUILD.bazel b/Compiler/ProjectSystem/Bridge/Tests/BUILD.bazel new file mode 100644 index 00000000..ea9ad500 --- /dev/null +++ b/Compiler/ProjectSystem/Bridge/Tests/BUILD.bazel @@ -0,0 +1,7 @@ +load("@rules_cc//cc:cc_binary.bzl", "cc_binary") + +cc_binary( + name = "project_registry_tests", + srcs = ["RegistryTests.cpp"], + deps = ["//Compiler/ProjectSystem/Bridge:project_driver", "@catch3//:catch2_main", "@catch3//src/Progmasoft:catch3"], +) diff --git a/Compiler/ProjectSystem/Bridge/Tests/RegistryTests.cpp b/Compiler/ProjectSystem/Bridge/Tests/RegistryTests.cpp new file mode 100644 index 00000000..9c53e4ff --- /dev/null +++ b/Compiler/ProjectSystem/Bridge/Tests/RegistryTests.cpp @@ -0,0 +1,91 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include + +#include "Compiler/ProjectSystem/Bridge/ProjectDriver.hpp" + +namespace +{ + auto + Records() -> std::vector + { + return { "visual-xsharp-sources-v6", + "0.4.0", + "default", + "llvm", + "debug", + "all", + "true", + "false", + "false", + "false", + "true", + "true", + "true", + "0", + "aot", + "none", + "build/debug", + "0", + "1", + "0", + "0", + "Application", + "Example.Program", + "Sources", + "0" }; + } + auto + Frame(const std::vector &records) -> std::string + { + std::string framed; + for (const auto &record : records) + { + framed.append(record); + framed.push_back('\0'); + } + return framed; + } +} // namespace + +TEST_CASE( + "project registry copies valid source records and rejects oversized counts") +{ + namespace Driver = Visual::XSharp::Driver; + auto records = Records(); + auto framed = Frame(records); + const auto project = Driver::ParseProjectRegistry(framed, true); + REQUIRE(project); + CHECK(project->executables.size() == 1U); + CHECK(project->entry == "Example.Program"); + framed.assign(framed.size(), 'x'); + CHECK(project->entry == "Example.Program"); + for (const auto index : { 17U, 18U, 19U, 20U, 24U }) + { + auto oversized = records; + oversized[index] = "18446744073709551615"; + CHECK_FALSE(Driver::ParseProjectRegistry(Frame(oversized), false)); + oversized[index] = "1000000"; + CHECK_FALSE(Driver::ParseProjectRegistry(Frame(oversized), false)); + } +} + +TEST_CASE("project registry validates framing and source requirements before " + "allocation") +{ + namespace Driver = Visual::XSharp::Driver; + CHECK_FALSE(Driver::ParseProjectRegistry({}, false)); + std::string oversized(4U * 1024U * 1024U + 1U, '\0'); + CHECK_FALSE(Driver::ParseProjectRegistry(oversized, false)); + auto records = Records(); + records.resize(21U); + records[18U] = "0"; + auto framed = Frame(records); + CHECK(Driver::ParseProjectRegistry(framed, false)); + CHECK_FALSE(Driver::ParseProjectRegistry(framed, true)); + framed.pop_back(); + CHECK_FALSE(Driver::ParseProjectRegistry(framed, false)); +} diff --git a/Compiler/Runtime/AARC/Fuzzing/BUILD.bazel b/Compiler/Runtime/AARC/Fuzzing/BUILD.bazel new file mode 100644 index 00000000..7aa86727 --- /dev/null +++ b/Compiler/Runtime/AARC/Fuzzing/BUILD.bazel @@ -0,0 +1,7 @@ +load("@rules_cc//cc:cc_binary.bzl", "cc_binary") + +cc_binary( + name = "ownership_fuzzer", + srcs = ["OwnershipFuzzer.cpp"], + deps = ["//Compiler/Runtime/AARC:aarc", "@llvm//:llvm"], +) diff --git a/Compiler/Runtime/AARC/Fuzzing/OwnershipFuzzer.cpp b/Compiler/Runtime/AARC/Fuzzing/OwnershipFuzzer.cpp new file mode 100644 index 00000000..ac2b729a --- /dev/null +++ b/Compiler/Runtime/AARC/Fuzzing/OwnershipFuzzer.cpp @@ -0,0 +1,96 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Runtime/AARC.hpp" + +namespace +{ + namespace Aarc = Visual::XSharp::Runtime::Aarc; + constexpr std::uint64_t kMarker = 0xABCDEF123456ULL; + struct Payload final + { + std::atomic_uint32_t *destructions{}; + std::uint64_t marker{}; + }; + void + Destroy(void *object) noexcept + { + auto *payload = static_cast(object); + payload->marker = 0U; + payload->destructions->fetch_add(1U, std::memory_order_relaxed); + } + const Aarc::TypeMetadata kMetadata{ Aarc::kAbiVersion, + 0U, + Aarc::TypeIdentity("Fuzz.Payload"), + sizeof(Payload), + alignof(Payload), + Destroy, + "Fuzz.Payload" }; +} // namespace + +extern "C" int +LLVMFuzzerTestOneInput(const std::uint8_t *data, std::size_t size) +{ + std::atomic_uint32_t destructions{}; + auto *payload = static_cast(Aarc::Allocate(kMetadata)); + if (payload == nullptr) + llvm::report_fatal_error("ownership fuzz payload allocation failed"); + payload->destructions = &destructions; + payload->marker = kMarker; + const auto weak = Aarc::MakeWeak(payload); + const auto unowned = Aarc::MakeUnowned(payload); + std::atomic_bool start{}; + std::atomic_bool corrupt{}; + const auto workerCount + = size == 0U ? 2U : 1U + static_cast(data[0] % 4U); + const auto iterations + = size < 2U ? 8U : 1U + static_cast(data[1] % 64U); + std::vector workers; + for (unsigned worker = 0U; worker < workerCount; ++worker) + { + const auto localWeak = Aarc::CopyWeak(weak); + const auto localUnowned = Aarc::CopyUnowned(unowned); + workers.emplace_back([&, localWeak, localUnowned, worker] { + while (!start.load(std::memory_order_acquire)) + std::this_thread::yield(); + for (unsigned iteration = 0U; iteration < iterations; ++iteration) + { + auto *locked = static_cast( + (iteration + worker) % 2U == 0U + ? Aarc::LockWeak(localWeak) + : Aarc::LoadUnowned(localUnowned)); + if (locked != nullptr) + { + // Reads occur only while a temporary strong owner protects + // the payload. Destructor writes must never overlap them. + if (locked->marker != kMarker) + corrupt.store(true, std::memory_order_relaxed); + Aarc::ReleaseStrong(locked); + } + const auto extra = Aarc::CopyWeak(localWeak); + Aarc::ReleaseWeak(extra); + std::this_thread::yield(); + } + Aarc::ReleaseWeak(localWeak); + Aarc::ReleaseUnowned(localUnowned); + }); + } + start.store(true, std::memory_order_release); + Aarc::ReleaseStrong(payload); + workers.clear(); // jthread joins before the independent destructor oracle. + if (corrupt.load() || destructions.load() != 1U + || Aarc::LockWeak(weak) != nullptr + || Aarc::LoadUnowned(unowned) != nullptr) + llvm::report_fatal_error( + "ownership race resurrected or multiply destroyed a payload"); + Aarc::ReleaseWeak(weak); + Aarc::ReleaseUnowned(unowned); + return 0; +} diff --git a/Documents/CORE-IR.md b/Documents/CORE-IR.md index 3ac3916c..5ae93a33 100644 --- a/Documents/CORE-IR.md +++ b/Documents/CORE-IR.md @@ -241,6 +241,36 @@ moving evaluation across a `break`, `continue`, or return. CorePrep consumes the verified structure and materializes loop headers, exits, latches, and the distinct `for` update block. +The Haskell CorePrep lowering and the native Core-to-CorePrep adapter must +build the same loop shape: + +- a `while` or `for` condition owns a dedicated header block. The statements + that precede the loop stay in the incoming block, which jumps to the header + once; every back-edge targets the header, never the incoming block; +- a numeric condition's canonicalizing `value != 0` comparison belongs to that + header and is re-evaluated on every iteration; +- the `for` update region is entered by normal body completion and by + `continue`, and every open tail of the update region jumps to the header. + The region's own entry is its `continue` target but is not its successor; +- `&&` and `||` are control flow, not eager two-operand instructions. A + Boolean result slot is initialized with the short-circuit value, the left + operand selects a branch, and only the block on the evaluating edge computes + the right operand and overwrites the slot. Calls, traps, and non-termination + in the right operand therefore stay conditional. A short-circuit condition + of a loop is evaluated starting at the loop header, so the loop's own + body/exit branch may sit in the operator's join block. + +`LoopLoweringTests.cpp` and `ShortCircuitLoweringTests.cpp` under +`Compiler/Core/Tests/` assert these edges exactly on the native adapter. +`LoopExecutionTests.cpp` and `ShortCircuitExecutionTests.cpp` under +`Compiler/Backend/LLVM/Tests/` execute each form through CorePrep, Xpp, Xmm, +and LLVM with both native optimizer settings and compare the result with host +code; the short-circuit programs guard a division or a recursive call, so an +eager right operand traps or never returns instead of merely producing the +same Boolean. A CorePrep, Xpp, +or Xmm verifier cannot reject a wrong back-edge by itself: a block that jumps +to itself is a well-formed control-flow graph. + ## Expressions Every Core expression has a statically queryable type. diff --git a/Documents/DEVELOPER-RECIPES.md b/Documents/DEVELOPER-RECIPES.md index dc47afc6..63f7e604 100644 --- a/Documents/DEVELOPER-RECIPES.md +++ b/Documents/DEVELOPER-RECIPES.md @@ -17,6 +17,7 @@ just test just benchmark just fuzz just fuzz-stress +just fuzz-thread # TSan run of threaded fuzz targets on a supported host just sanitize # ASan + UBSan just sanitize-thread # Separate TSan run on a supported host just verify-helpers diff --git a/Documents/FUZZING.md b/Documents/FUZZING.md index 3a07dec6..b8fc37c2 100644 --- a/Documents/FUZZING.md +++ b/Documents/FUZZING.md @@ -15,27 +15,52 @@ memory safety or complete language coverage. | `lexer_fuzzer` | Arbitrary bytes through the frontend lexer ABI; complete token/diagnostic evaluation | Native ABI bridge, not GHC-generated lexer branches | | `parser_fuzzer` | Arbitrary bytes through syntax analysis; complete AST/diagnostic evaluation | Native ABI bridge, not GHC-generated parser branches | | `source_llvm_fuzzer` | Arbitrary source through Core/CorePrep, Xpp/Xmm verification and LLVM lowering | First-party C++ pipeline | -| `differential_fuzzer` | Generated arithmetic compiled with native optimizers disabled/enabled and compared with an independent evaluator | First-party C++ pipeline and JIT bridge | - -The differential oracle independently evaluates bounded generated arithmetic, -compiles its source once, lowers the same verified Core with Xpp/Xmm +| `differential_fuzzer` | Generated arithmetic and control flow compiled with native optimizers disabled/enabled and compared with an independent evaluator | First-party C++ pipeline and JIT bridge | +| `cli_fuzzer` | NUL-separated arguments through the typed command-line parser; two parses must produce equal typed models, leave `argv` unchanged, attach a diagnostic to every rejection and select a command on every acceptance | First-party C++ CLI parser | +| `project_fuzzer` | Evaluator registry records, both raw and as one mutated field of a valid document; decoding must be deterministic and a source-requiring decode may only accept what the permissive decode accepts | First-party C++ registry decoder; no evaluator process or filesystem access | +| `repl_fuzzer` | Up to eight operations on one persistent session: arithmetic cells checked against an independent accumulator, type queries that must not change history, failed-cell rollback and reset | First-party C++ session and JIT bridge | +| `ownership_fuzzer` | One to four threads racing weak locks, unowned loads and weak copies against the final strong release; the payload must be destroyed exactly once, never resurrected and never read after destruction | AARC runtime | +| `frontend-fuzz` | Lexer, parser and source-to-CorePrep stages mutated in-process with GHC HPC tick feedback | GHC-compiled frontend modules; no native sanitizer | + +The differential oracle independently evaluates one generated program per +input: a bounded arithmetic expression, a classic `for` loop with `continue` +and `break`, a `do`/`while` loop, an `if`/`else` over that expression, a +`while` loop preceded by its own initializers, or a recursion that terminates +only because `||` and `&&` skip their right operands. The expected value is computed +by ordinary host code in the harness, never by a second compiler path. It +compiles the source once, lowers the same verified Core with Xpp/Xmm optimizations both disabled and enabled, executes both verified -artifacts through ORC, and compares all three results. This detects miscompiles -within that generated subset; it is not an oracle for arbitrary Visual X# +artifacts through ORC, and compares all three results. Agreement between the +two optimizer settings is not accepted by itself: both consume the same +Core-to-CorePrep adapter, so a defect there makes them agree on the same wrong +or non-terminating program. This detects miscompiles within that generated +subset; it is not an oracle for arbitrary Visual X# programs. Invalid source is a normal rejection, whereas internal failures and verified-model inconsistencies fail the campaign. Arbitrary source and generated arithmetic use separate corpora and equal per-target time budgets. This lets source mutations reach native lowering without repeatedly creating two ORC sessions for unrelated generated code. -The differential generator consumes at most 31 selector bytes; its 64-byte -input limit keeps mutations near the bytes that influence the program. +The differential generator consumes at most 33 bytes: 31 expression selectors, +one shape byte and one trip-count byte. Its 64-byte input limit keeps +mutations near the bytes that influence the program. GHC frontend code and prebuilt LLVM dependencies do not receive Clang native -coverage instrumentation. Lexer/parser execution must not be presented as -coverage-guided exploration of their Haskell implementation. Haskell tests and -HPC coverage remain complementary gates, not substitutes for a frontend-specific -feedback-guided engine. +coverage instrumentation. Running `lexer_fuzzer` or `parser_fuzzer` must not be +presented as coverage-guided exploration of the Haskell implementation. + +`frontend-fuzz` is the separate feedback engine for that code. It is a Cabal +executable built with `--enable-coverage` in its own `dist-fuzz-coverage` +build directory, so it never shares package state with the ordinary frontend +build. It resets the HPC counters before each input, keeps an input when it +reaches a tick of a `Visual.XSharp.*` module that no earlier input reached, +and refuses to run when the production modules carry no ticks. Mutations are +deterministic for a recorded seed, inputs are limited to 8192 bytes and five +seconds, and the runtime heap is limited to 512 MiB. A Haskell exception or a +timeout writes the exact input as `failure.seed` and fails the campaign; +`frontend-fuzz STAGE --replay FILE` replays it. This engine has no sanitizer +and reports tick coverage, which is not comparable with libFuzzer edge +coverage. ## Run a campaign @@ -62,7 +87,11 @@ Each campaign has a 30-second per-input timeout and a finite input length: | Lexer | 65536 | 1024 | | Parser | 65536 | 1536 | | Source/LLVM | 65536 | 4096 | -| Differential arithmetic | 64 | 4096 | +| Differential | 64 | 4096 | +| CLI | 16384 | 768 | +| Project registry | 65536 | 768 | +| REPL | 64 | 4096 | +| Ownership | 64 | 768 | ASan intentionally retains freed allocations in quarantine. Fuzz-only settings bound this cache to 64 MiB, with a 256 KiB thread-local cache; the nonzero @@ -73,8 +102,9 @@ normal quarantine settings. ## Corpus synchronization and reports Wire seeds come from production writers, so format-version changes do not leave -handwritten supposedly valid documents behind. Lexer, parser, source and differential seeds -come from `Compiler/Fuzzing/Corpus/`. Set `VXS_FUZZ_CORPUS` to retain mutation +handwritten supposedly valid documents behind. Every other target has +versioned seeds under `Compiler/Fuzzing/Corpus//`; a target without +that directory fails the campaign instead of starting from nothing. Set `VXS_FUZZ_CORPUS` to retain mutation corpora across local runs. Updated versioned seeds are added without overwriting older discovered inputs; conflicting contents under a stable hash fail closed. @@ -106,8 +136,31 @@ whole corpus and sanitizer cache rather than just the final input. `wire_fuzz_smoke` exercises 1024 deterministic mutations of each of four valid documents. It runs independently of libFuzzer and does not claim guided -coverage. `source_fuzz_smoke` checks valid-source lowering and the differential -oracle before mutation campaigns begin. +coverage. `source_fuzz_smoke` checks valid-source lowering and then runs the +differential oracle on every generated program shape at trip counts 0 through +11 before mutation campaigns begin. Set `VXS_FUZZ_TRACE=1` to print each +generated source with its reference and optimized LLVM IR. + +Neither smoke program nor the HPC engine has libFuzzer's per-input timeout, and +a miscompiled generated loop does not return. The developer helper therefore +bounds every smoke, libFuzzer and HPC process as a whole: 90 seconds for a +smoke program, the campaign duration plus 90 seconds for a libFuzzer target and +plus 300 seconds for the HPC engine. On expiry it terminates the complete +process tree and reports a watchdog failure. Build tools are not bounded by +this watchdog. Do not run a smoke binary directly without an external time +limit when investigating a suspected hang. + +## ThreadSanitizer campaign + +AddressSanitizer and ThreadSanitizer cannot share one executable. +`go run ./helpers/cmd/develop fuzz-thread` builds only the targets that start +their own threads, currently `ownership_fuzzer`, with libFuzzer and +ThreadSanitizer, proves that the runtime reports an intentional data race, and +then runs a bounded campaign with the same corpus, limits and report checks. +The command fails on Windows, where Clang has no ThreadSanitizer runtime; it +never reports a skipped run as success. The fuzzing workflow runs it on macOS +and native Ubuntu. The Fedora container job does not run it for the +shadow-memory reason given below. Compiler Tier 1/2/3 run complete native suites with ASan/UBSan. Tier 1 macOS and Tier 2 native Ubuntu additionally run separate TSan suites. Fedora Tier 3 @@ -123,8 +176,22 @@ protection must require these actual checks. ## Remaining coverage boundaries -CLI argument generation, project/lockfile inputs, persistent REPL sessions and -broader generated language programs need independent oracles. Deep semantic -cases, ownership concurrency and frontend feedback-guided coverage are not -established by the five existing targets. Expand these deliberately instead of -equating a green workflow with completion of the entire security program. +The harnesses above are evidence about the inputs they ran, not proofs: + +- the CLI, project and REPL oracles check determinism, rollback and a small + arithmetic model. They do not model every option interaction, lockfile + refresh, project evaluation or REPL declaration form; +- the differential generator covers integer arithmetic, three loop forms, + one conditional and one guarded recursion. Closures, other scalar types, + ownership and templates have no generated-program oracle here; their + executable checks live in the component test suites; +- `ownership_fuzzer` varies thread count and iteration count, not arbitrary + interleavings, and its ThreadSanitizer campaign does not run on Windows or in + the Fedora container; +- HPC feedback is expression-tick coverage of the frontend, without + memory-safety instrumentation, and its mutator is byte- and token-level, not + grammar-aware; +- prebuilt LLVM and the GHC runtime are not instrumented by any campaign. + +Expand these deliberately instead of equating a green workflow with completion +of the entire security program. diff --git a/Documents/TESTING.md b/Documents/TESTING.md index b92f2f80..f35a3205 100644 --- a/Documents/TESTING.md +++ b/Documents/TESTING.md @@ -118,7 +118,7 @@ go run ./helpers/cmd/develop doctor go run ./helpers/cmd/develop test ``` -The command executes 18 Catch3 binaries, one C11 ABI contract executable and +The command executes 19 Catch3 binaries, one C11 ABI contract executable and one source/differential fuzz smoke executable on Windows 10/11, macOS Sequoia/Tahoe, Ubuntu 26.04 LTS, and Fedora 43. Bazel selects the host configuration automatically; no public test instruction diff --git a/Interactive/Fuzzing/BUILD.bazel b/Interactive/Fuzzing/BUILD.bazel new file mode 100644 index 00000000..ac6de162 --- /dev/null +++ b/Interactive/Fuzzing/BUILD.bazel @@ -0,0 +1,7 @@ +load("@rules_cc//cc:cc_binary.bzl", "cc_binary") + +cc_binary( + name = "repl_fuzzer", + srcs = ["ReplFuzzer.cpp"], + deps = ["//Interactive:interactive_runtime", "@llvm//:llvm"], +) diff --git a/Interactive/Fuzzing/ReplFuzzer.cpp b/Interactive/Fuzzing/ReplFuzzer.cpp new file mode 100644 index 00000000..8bb49415 --- /dev/null +++ b/Interactive/Fuzzing/ReplFuzzer.cpp @@ -0,0 +1,70 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Interactive/Session.hpp" + +extern "C" int +LLVMFuzzerTestOneInput(const std::uint8_t *data, std::size_t size) +{ + namespace Repl = Visual::XSharp::Interactive; + Repl::Session session; + std::int64_t expected{}; + bool hasPrevious{}; + const auto initial = session.Evaluate("return 0;"); + if (initial.status != Repl::CellStatus::Value) + llvm::report_fatal_error( + "REPL fuzz runtime could not evaluate its valid initial cell"); + hasPrevious = true; + for (std::size_t index = 0U; index < std::min(size, std::size_t{ 8U }); + ++index) + { + const auto selector = data[index]; + if (selector % 5U == 0U) + { + if (session.Reset() || !session.History().empty()) + llvm::report_fatal_error( + "REPL reset retained failed resources or history"); + hasPrevious = false; + expected = 0; + continue; + } + if (selector % 5U == 1U) + { + const auto history = session.History(); + if (session.Evaluate("return MissingFuzzName;").status + != Repl::CellStatus::Error + || session.History() != history) + llvm::report_fatal_error( + "REPL failed-cell rollback changed history"); + continue; + } + const auto increment = static_cast(selector % 11U); + const auto expression = std::string("return ") + + (hasPrevious ? "vxsiPrevious + " : "") + + std::to_string(increment) + ";"; + const auto history = session.History(); + if (session.TypeOf(expression).status != Repl::CellStatus::Type + || session.History() != history) + llvm::report_fatal_error("REPL type query mutated session state"); + expected += increment; + const auto value = session.Evaluate(expression); + const auto *integer + = value.value ? std::get_if(&value.value->payload) + : nullptr; + if (value.status != Repl::CellStatus::Value || integer == nullptr + || *integer != expected) + llvm::report_fatal_error("persistent REPL session disagrees with " + "independent arithmetic state"); + hasPrevious = true; + } + if (session.Reset()) + llvm::report_fatal_error("REPL fuzz resource teardown failed"); + return 0; +} diff --git a/helpers/internal/development/build.go b/helpers/internal/development/build.go index e12a77a0..c89c7206 100644 --- a/helpers/internal/development/build.go +++ b/helpers/internal/development/build.go @@ -33,6 +33,7 @@ var nativeTargets = []string{ "//Compiler/Runtime/AARC/Tests:aarc_runtime_tests", "//Compiler/Runtime/AARC/Tests:aarc_c_abi_tests", "//Compiler/Fuzzing:source_fuzz_smoke", + "//Compiler/ProjectSystem/Bridge/Tests:project_registry_tests", "//Interactive/Tests:interactive_tests", } diff --git a/helpers/internal/development/clean.go b/helpers/internal/development/clean.go index f8ba7462..aee2023d 100644 --- a/helpers/internal/development/clean.go +++ b/helpers/internal/development/clean.go @@ -12,6 +12,7 @@ import ( var generatedBuildPaths = []string{ "Compiler/dist-newstyle", + "Compiler/dist-fuzz-coverage", "ProjectSystem/.gradle", "ProjectSystem/build", "Analyzer/.gradle", diff --git a/helpers/internal/development/cli.go b/helpers/internal/development/cli.go index d9b1b398..edccb5b2 100644 --- a/helpers/internal/development/cli.go +++ b/helpers/internal/development/cli.go @@ -39,7 +39,8 @@ func newCommand(runner commandRunner, output, errorOutput io.Writer) *cobra.Comm {"test", "Build and execute every native contract suite.", 0, true}, {"sanitize", "Run address, undefined, address-undefined, or thread sanitizer suites.", 1, true}, {"version", "Validate major.minor.patch[.revision] release metadata.", 1, false}, - {"fuzz", "Run short wire, lexer, parser, and LLVM-source campaigns.", 0, false}, + {"fuzz", "Run bounded ASan/UBSan libFuzzer campaigns and the Haskell HPC campaign.", 0, false}, + {"fuzz-thread", "Run the threaded fuzz targets under ThreadSanitizer (macOS/Linux).", 0, false}, {"incremental-clean-build", "Rebuild while preserving downloaded dependencies.", 0, false}, {"cold-clean-build", "Expunge Bazel state and rebuild the compiler.", 0, false}, {"clean", "Remove generated Bazel, Cabal, and Gradle output.", 0, false}, diff --git a/helpers/internal/development/cli_test.go b/helpers/internal/development/cli_test.go index 7e9be798..95ea1202 100644 --- a/helpers/internal/development/cli_test.go +++ b/helpers/internal/development/cli_test.go @@ -60,12 +60,12 @@ func TestFuzzBuildPlansSeparateSmokeAndRuntimeMain(t *testing.T) { } drivers := 0 for _, argument := range campaign { - if strings.HasPrefix(argument, "//Compiler/Fuzzing:") { + if strings.HasPrefix(argument, "//") { drivers++ } } - if drivers != 5 { - t.Fatalf("expected five campaign drivers: %v", campaign) + if drivers != 9 { + t.Fatalf("expected nine campaign drivers: %v", campaign) } if sanitizer != "" { for _, plan := range [][]string{smoke, campaign} { diff --git a/helpers/internal/development/commands.go b/helpers/internal/development/commands.go index 523ae4f8..f8bc7e77 100644 --- a/helpers/internal/development/commands.go +++ b/helpers/internal/development/commands.go @@ -78,6 +78,14 @@ func executeWorkflow(arguments []string, runner commandRunner) error { return err } return runFuzzCampaign(repository, currentHost, runner, false, true) + case "fuzz-thread": + if len(commandArguments) != 0 || len(bazelArguments) != 0 { + return errors.New("fuzz-thread does not accept arguments") + } + if err := requireBuildTools(currentHost, runner); err != nil { + return err + } + return runThreadFuzzCampaign(repository, currentHost, runner) case "fuzz-stress": if len(bazelArguments) != 0 || (len(commandArguments) != 0 && !(len(commandArguments) == 1 && strings.EqualFold(commandArguments[0], "--asan"))) { return errors.New("fuzz-stress accepts only the optional --asan flag") diff --git a/helpers/internal/development/fuzz.go b/helpers/internal/development/fuzz.go index 0ede95a6..17206f6f 100644 --- a/helpers/internal/development/fuzz.go +++ b/helpers/internal/development/fuzz.go @@ -53,12 +53,9 @@ func fuzzBuildArguments(configuration, sanitizerConfiguration, macRuntime string campaign = append(campaign, "--linkopt="+macRuntime) } smoke = append(smoke, "//Compiler/Fuzzing:wire_fuzz_smoke", "//Compiler/Fuzzing:source_fuzz_smoke") - campaign = append(campaign, - "//Compiler/Fuzzing:wire_fuzzer", - "//Compiler/Fuzzing:lexer_fuzzer", - "//Compiler/Fuzzing:parser_fuzzer", - "//Compiler/Fuzzing:source_llvm_fuzzer", - "//Compiler/Fuzzing:differential_fuzzer") + for _, target := range nativeFuzzTargets() { + campaign = append(campaign, target.label) + } return smoke, campaign } @@ -109,8 +106,8 @@ func runFuzzCampaign(repository string, currentHost host, runner commandRunner, if err := os.Mkdir(artifacts, 0o700); err != nil { return fmt.Errorf("could not create fuzz artifact directory %q: %w", work, err) } - stageNames := []string{"wire", "lexer", "parser", "source", "differential"} - for _, stage := range stageNames { + for _, target := range nativeFuzzTargets() { + stage := target.corpus corpus := filepath.Join(corpusRoot, stage) if err := os.MkdirAll(corpus, 0o700); err != nil { return fmt.Errorf("could not prepare %s corpus; preserved %q: %w", stage, work, err) @@ -174,65 +171,24 @@ func runFuzzCampaign(repository string, currentHost host, runner commandRunner, if err := runner.Run(repository, nil, bazel, campaignArguments...); err != nil { return fmt.Errorf("could not build instrumented fuzz targets; preserved %q: %w", work, err) } - targets := []struct { - binary string - corpus string - maxLength string - rssLimit string - }{ - {"wire_fuzzer", "wire", "16384", "768"}, - {"lexer_fuzzer", "lexer", "65536", "1024"}, - {"parser_fuzzer", "parser", "65536", "1536"}, - {"source_llvm_fuzzer", "source", "65536", "4096"}, - // The depth-four generator consumes at most 31 selector bytes. Bound - // mutations close to this semantic input rather than evolving unused - // tails alongside the expensive two-module execution oracle. - {"differential_fuzzer", "differential", "64", "4096"}, - } - var records []map[string]any - for _, target := range targets { - fuzzer := filepath.Join(repository, "bazel-bin", "Compiler", "Fuzzing", target.binary+currentHost.executable) - if target.binary != "wire_fuzzer" { - if err := copyFile(frontendLibrary, filepath.Join(filepath.Dir(fuzzer), filepath.Base(frontendLibrary)), 0o755); err != nil { - return fmt.Errorf("could not stage frontend for %s; preserved %q: %w", target.binary, work, err) - } - } - campaignArtifacts := filepath.Join(artifacts, target.binary) - if err := os.MkdirAll(campaignArtifacts, 0o700); err != nil { - return fmt.Errorf("could not create %s artifact directory: %w", target.binary, err) - } - arguments := []string{ - filepath.Join(corpusRoot, target.corpus), - "-max_total_time=" + strconv.Itoa(duration), - "-max_len=" + target.maxLength, - "-timeout=30", - "-rss_limit_mb=" + target.rssLimit, - "-use_value_profile=1", - "-verbosity=0", - "-print_final_stats=1", - "-artifact_prefix=" + campaignArtifacts + string(os.PathSeparator), - } - output, runErr := runner.RunWithInput(repository, selectedEnvironment, "", fuzzer, arguments...) - if err := os.WriteFile(filepath.Join(artifacts, target.binary+".log"), []byte(output), 0o600); err != nil { - return fmt.Errorf("could not preserve campaign log: %w", err) - } - fmt.Print(output) - statistics, statisticsErr := parseFuzzStatistics(output) - if runErr == nil && statisticsErr != nil { - runErr = statisticsErr - } - records = append(records, map[string]any{"target": target.binary, "seconds": duration, "rss_limit_mb": target.rssLimit, "sanitizer": sanitizerConfiguration, "native_coverage": true, "haskell_native_coverage": false, "statistics": statistics, "success": runErr == nil}) - report, err := json.MarshalIndent(records, "", " ") - if err != nil { - return err - } - if err := os.WriteFile(filepath.Join(artifacts, "campaigns.json"), report, 0o600); err != nil { + campaign := fuzzCampaign{ + repository: repository, + corpusRoot: corpusRoot, + artifacts: artifacts, + work: work, + report: "campaigns.json", + duration: duration, + environment: selectedEnvironment, + sanitizer: sanitizerConfiguration, + executable: currentHost.executable, + } + for _, target := range nativeFuzzTargets() { + if err := campaign.run(runner, target, frontendLibrary); err != nil { return err } - if runErr != nil { - return fmt.Errorf("%s campaign failed; corpus and crash artifacts preserved in %q: %w", target.binary, work, runErr) - } - fmt.Printf("%s campaign completed (%d seconds; RSS <= %s MiB).\n", target.binary, duration, target.rssLimit) + } + if err := runHaskellFuzz(repository, corpusRoot, artifacts, duration, runner); err != nil { + return fmt.Errorf("Haskell feedback campaign failed; preserved %q: %w", work, err) } if os.Getenv("CI") == "true" { fmt.Printf("Campaign logs and coverage limits preserved in %s.\n", work) @@ -254,6 +210,67 @@ func runFuzzCampaign(repository string, currentHost host, runner commandRunner, return nil } +// fuzzCampaign carries the settings shared by every libFuzzer target of one +// helper invocation and accumulates their structured report. +type fuzzCampaign struct { + repository, corpusRoot, artifacts, work, report string + duration int + environment []string + sanitizer, executable string + records []map[string]any +} + +// run executes one instrumented target against its persistent corpus. The +// report is rewritten after every target so a later failure still leaves the +// earlier measurements, and a process that exits zero without libFuzzer's final +// counters is a failure rather than an unexplained success. +func (campaign *fuzzCampaign) run(runner commandRunner, target fuzzTarget, frontendLibrary string) error { + packagePath, _, _ := strings.Cut(strings.TrimPrefix(target.label, "//"), ":") + fuzzer := filepath.Join(campaign.repository, "bazel-bin", filepath.FromSlash(packagePath), target.binary+campaign.executable) + if target.frontend { + if err := copyFile(frontendLibrary, filepath.Join(filepath.Dir(fuzzer), filepath.Base(frontendLibrary)), 0o755); err != nil { + return fmt.Errorf("could not stage frontend for %s; preserved %q: %w", target.binary, campaign.work, err) + } + } + campaignArtifacts := filepath.Join(campaign.artifacts, target.binary) + if err := os.MkdirAll(campaignArtifacts, 0o700); err != nil { + return fmt.Errorf("could not create %s artifact directory: %w", target.binary, err) + } + arguments := []string{ + filepath.Join(campaign.corpusRoot, target.corpus), + "-max_total_time=" + strconv.Itoa(campaign.duration), + "-max_len=" + target.maxLength, + "-timeout=30", + "-rss_limit_mb=" + target.rssLimit, + "-use_value_profile=1", + "-verbosity=0", + "-print_final_stats=1", + "-artifact_prefix=" + campaignArtifacts + string(os.PathSeparator), + } + output, runErr := runner.RunWithInput(campaign.repository, campaign.environment, "", fuzzer, arguments...) + if err := os.WriteFile(filepath.Join(campaign.artifacts, target.binary+".log"), []byte(output), 0o600); err != nil { + return fmt.Errorf("could not preserve campaign log: %w", err) + } + fmt.Print(output) + statistics, statisticsErr := parseFuzzStatistics(output) + if runErr == nil && statisticsErr != nil { + runErr = statisticsErr + } + campaign.records = append(campaign.records, map[string]any{"target": target.binary, "seconds": campaign.duration, "rss_limit_mb": target.rssLimit, "sanitizer": campaign.sanitizer, "native_coverage": true, "haskell_native_coverage": false, "statistics": statistics, "success": runErr == nil}) + report, err := json.MarshalIndent(campaign.records, "", " ") + if err != nil { + return err + } + if err := os.WriteFile(filepath.Join(campaign.artifacts, campaign.report), report, 0o600); err != nil { + return err + } + if runErr != nil { + return fmt.Errorf("%s campaign failed; corpus and crash artifacts preserved in %q: %w", target.binary, campaign.work, runErr) + } + fmt.Printf("%s campaign completed (%d seconds; RSS <= %s MiB; %s).\n", target.binary, campaign.duration, target.rssLimit, campaign.sanitizer) + return nil +} + func fuzzDuration(stress bool, configured string) (int, error) { if configured == "" { if stress { diff --git a/helpers/internal/development/fuzz_haskell.go b/helpers/internal/development/fuzz_haskell.go new file mode 100644 index 00000000..b2ea1f2b --- /dev/null +++ b/helpers/internal/development/fuzz_haskell.go @@ -0,0 +1,138 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "encoding/json" + "fmt" + "os" + "path/filepath" + "strconv" + "strings" +) + +// The native C ABI wrapper cannot feed back GHC branches. This executable is +// rebuilt in an isolated HPC tree and retains inputs that hit new production +// ticks, rather than pretending native callback coverage represents Haskell. +func runHaskellFuzz(repository, corpusRoot, artifacts string, duration int, runner commandRunner) error { + directory := filepath.Join(repository, "Compiler") + options := []string{"--enable-coverage", "--disable-tests", "--builddir=dist-fuzz-coverage"} + build := append([]string{"build", "exe:frontend-fuzz"}, options...) + if err := runner.Run(directory, nil, "cabal", build...); err != nil { + return fmt.Errorf("HPC frontend build failed: %w", err) + } + arguments := append([]string{"list-bin", "exe:frontend-fuzz"}, options...) + binary, err := runner.OutputIn(directory, "cabal", arguments...) + if err != nil || !filepath.IsAbs(binary) || strings.ContainsAny(binary, "\r\n") { + return fmt.Errorf("could not resolve HPC frontend executable: %q (%v)", binary, err) + } + for _, stage := range []string{"lexer", "parser", "source"} { + corpus := filepath.Join(corpusRoot, "haskell-"+stage) + resultPath := filepath.Join(artifacts, "haskell-"+stage) + for _, path := range []string{corpus, resultPath} { + if err := os.MkdirAll(path, 0o700); err != nil { + return err + } + } + if err := syncSeedCorpus(filepath.Join(repository, "Compiler", "Fuzzing", "Corpus", stage), corpus); err != nil { + return err + } + // An HPC executable loads an existing tick file at startup and aborts + // when its module hashes belong to an earlier build. Give every stage a + // fresh file beside its report instead of the default one in the + // working directory, which would also leave output in the source tree. + tickFile := filepath.Join(resultPath, "frontend-fuzz.tix") + if err := os.Remove(tickFile); err != nil && !os.IsNotExist(err) { + return fmt.Errorf("could not reset the HPC tick file: %w", err) + } + // The engine writes its report to this file only after a completed + // campaign. Remove an earlier one so a crashed run cannot inherit it. + reportFile := filepath.Join(resultPath, "campaign.txt") + if err := os.Remove(reportFile); err != nil && !os.IsNotExist(err) { + return fmt.Errorf("could not reset the HPC campaign report: %w", err) + } + output, runErr := runner.RunWithInput(directory, []string{"HPCTIXFILE=" + tickFile}, "", binary, stage, strconv.Itoa(duration), corpus, resultPath, "12345") + fmt.Print(output) + if err := os.WriteFile(filepath.Join(resultPath, "output.log"), []byte(output), 0o600); err != nil { + return err + } + stats, statsErr := readHaskellFuzzReport(reportFile, output, stage) + record := map[string]any{"stage": stage, "seconds": duration, "coverage_engine": "ghc-hpc", "asan_instrumented": false, "heap_limit_mib": 512, "statistics": stats, "success": runErr == nil && statsErr == nil} + report, err := json.MarshalIndent(record, "", " ") + if err != nil { + return err + } + if err := os.WriteFile(filepath.Join(resultPath, "campaign.json"), report, 0o600); err != nil { + return err + } + if runErr != nil { + return fmt.Errorf("HPC %s campaign failed; artifacts in %q: %w", stage, resultPath, runErr) + } + if statsErr != nil { + return statsErr + } + } + return nil +} + +// readHaskellFuzzReport validates the report file the engine wrote for this +// run. Captured process output interleaves stdout with stderr, where the GHC +// runtime may print its own diagnostics after the report; those lines are kept +// in the log but are not statistics. The output must still announce the report, +// so a stale or hand-placed file is never accepted without a completed run. +func readHaskellFuzzReport(reportFile, output, stage string) (map[string]uint64, error) { + if !strings.Contains(output, "HPC_FUZZ_RESULT") { + return nil, fmt.Errorf("HPC %s campaign omitted its coverage report", stage) + } + report, err := os.ReadFile(reportFile) + if err != nil { + return nil, fmt.Errorf("HPC %s campaign did not write its report file: %w", stage, err) + } + return parseHaskellFuzzStatistics("HPC_FUZZ_RESULT\n"+string(report), stage) +} + +func parseHaskellFuzzStatistics(output, stage string) (map[string]uint64, error) { + _, report, found := strings.Cut(output, "HPC_FUZZ_RESULT\n") + if !found { // Windows process output may have CRLF newlines. + _, report, found = strings.Cut(output, "HPC_FUZZ_RESULT\r\n") + } + if !found { + return nil, fmt.Errorf("HPC %s campaign omitted its coverage report", stage) + } + values := make(map[string]uint64) + stageSeen := false + for _, line := range strings.Split(strings.ReplaceAll(report, "\r\n", "\n"), "\n") { + if line == "" { + continue + } + key, value, found := strings.Cut(line, "=") + if !found { + return nil, fmt.Errorf("malformed HPC report line: %q", line) + } + if key == "stage" { + if stageSeen || value != stage { + return nil, fmt.Errorf("unexpected or duplicate HPC stage: %q", value) + } + stageSeen = true + continue + } + if _, duplicate := values[key]; duplicate { + return nil, fmt.Errorf("duplicate HPC statistic: %s", key) + } + number, err := strconv.ParseUint(value, 10, 64) + if err != nil { + return nil, fmt.Errorf("invalid HPC statistic %s: %w", key, err) + } + values[key] = number + } + if !stageSeen || values["executed_units"] == 0 || values["covered_ticks"] == 0 || values["covered_ticks"] > values["available_ticks"] { + return nil, fmt.Errorf("HPC %s campaign did not prove production coverage", stage) + } + for _, key := range []string{"new_units_added", "seed"} { + if _, found := values[key]; !found { + return nil, fmt.Errorf("HPC report omitted %s", key) + } + } + return values, nil +} diff --git a/helpers/internal/development/fuzz_haskell_report_test.go b/helpers/internal/development/fuzz_haskell_report_test.go new file mode 100644 index 00000000..7e5c9219 --- /dev/null +++ b/helpers/internal/development/fuzz_haskell_report_test.go @@ -0,0 +1,47 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "os" + "path/filepath" + "strings" + "testing" +) + +func TestHaskellReportFileIgnoresRuntimeDiagnosticsButNotMissingEvidence(t *testing.T) { + report := "stage=lexer\nexecuted_units=3255\ncovered_ticks=839\navailable_ticks=36407\nnew_units_added=33\nseed=12345\n" + // The GHC runtime may print to stderr after the report; captured output + // interleaves both streams. + output := "HPC_FUZZ_RESULT\r\n" + strings.ReplaceAll(report, "\n", "\r\n") + + "\r\nonIOComplete: failed to grab table semaphore (res=2439, err=-1), dropping request 0x6\n" + file := filepath.Join(t.TempDir(), "campaign.txt") + if err := os.WriteFile(file, []byte(report), 0o600); err != nil { + t.Fatal(err) + } + values, err := readHaskellFuzzReport(file, output, "lexer") + if err != nil || values["covered_ticks"] != 839 || values["executed_units"] != 3255 { + t.Fatalf("completed campaign rejected: %v / %v", values, err) + } + // The same trailing diagnostics are still not statistics. + if _, err := parseHaskellFuzzStatistics(output, "lexer"); err == nil { + t.Fatal("runtime diagnostics were accepted as campaign statistics") + } + // A report file without a run that announced it is stale evidence. + if _, err := readHaskellFuzzReport(file, "process exited successfully", "lexer"); err == nil { + t.Fatal("report file accepted without a completed campaign") + } + if _, err := readHaskellFuzzReport(filepath.Join(t.TempDir(), "missing.txt"), output, "lexer"); err == nil { + t.Fatal("missing report file accepted") + } + if _, err := readHaskellFuzzReport(file, output, "parser"); err == nil { + t.Fatal("report for another stage accepted") + } + if err := os.WriteFile(file, []byte(strings.Replace(report, "covered_ticks=839", "covered_ticks=0", 1)), 0o600); err != nil { + t.Fatal(err) + } + if _, err := readHaskellFuzzReport(file, output, "lexer"); err == nil { + t.Fatal("report without production coverage accepted") + } +} diff --git a/helpers/internal/development/fuzz_haskell_test.go b/helpers/internal/development/fuzz_haskell_test.go new file mode 100644 index 00000000..4ed0fab6 --- /dev/null +++ b/helpers/internal/development/fuzz_haskell_test.go @@ -0,0 +1,70 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "strings" + "testing" +) + +func TestHaskellReportRequiresActualProductionCoverage(t *testing.T) { + valid := "HPC_FUZZ_RESULT\nstage=lexer\nexecuted_units=3378\ncovered_ticks=958\navailable_ticks=36407\nnew_units_added=38\nseed=12345\n" + for _, report := range []string{valid, strings.ReplaceAll(valid, "\n", "\r\n")} { + values, err := parseHaskellFuzzStatistics(report, "lexer") + if err != nil || values["covered_ticks"] != 958 { + t.Fatalf("valid coverage rejected: %v / %v", values, err) + } + } + for _, bad := range []string{ + "process exited successfully", + strings.Replace(valid, "stage=lexer", "stage=parser", 1), + strings.Replace(valid, "executed_units=3378", "executed_units=0", 1), + strings.Replace(valid, "covered_ticks=958", "covered_ticks=0", 1), + strings.Replace(valid, "available_ticks=36407", "available_ticks=10", 1), + strings.Replace(valid, "seed=12345\n", "", 1), + valid + "seed=54321\n", + valid + "stage=lexer\n", + strings.Replace(valid, "new_units_added=38", "new_units_added=-1", 1), + } { + if _, err := parseHaskellFuzzStatistics(bad, "lexer"); err == nil { + t.Fatalf("missing/malformed coverage report accepted: %q", bad) + } + } +} + +func TestNativeFuzzInventoryIsComponentOwnedAndBounded(t *testing.T) { + labels, corpora := map[string]bool{}, map[string]bool{} + for _, target := range nativeFuzzTargets() { + if labels[target.label] || corpora[target.corpus] { + t.Fatalf("duplicate target/corpus: %+v", target) + } + labels[target.label], corpora[target.corpus] = true, true + if !strings.HasPrefix(target.label, "//") || !strings.HasSuffix(target.label, ":"+target.binary) || target.maxLength == "" || target.rssLimit == "" { + t.Fatalf("unbounded or disconnected target: %+v", target) + } + } + for _, label := range []string{"//Compiler/Cli/Fuzzing:cli_fuzzer", "//Compiler/ProjectSystem/Bridge/Fuzzing:project_fuzzer", "//Interactive/Fuzzing:repl_fuzzer", "//Compiler/Runtime/AARC/Fuzzing:ownership_fuzzer"} { + if !labels[label] { + t.Fatalf("missing component-local target %s", label) + } + } +} + +func TestFuzzProcessWatchdogDoesNotLimitBuildTools(t *testing.T) { + for _, item := range []struct { + name string + args []string + seconds int + }{ + {"source_fuzz_smoke.exe", nil, 90}, + {"wire_fuzzer", []string{"-max_total_time=30"}, 120}, + {"frontend-fuzz.exe", []string{"parser", "900"}, 1200}, + {"cabal", []string{"build", "all"}, 0}, + {"bazel", []string{"build", "//Compiler:vxs"}, 0}, + } { + if got := fuzzProcessSeconds(item.name, item.args); got != item.seconds { + t.Fatalf("%s timeout = %d; want %d", item.name, got, item.seconds) + } + } +} diff --git a/helpers/internal/development/fuzz_targets.go b/helpers/internal/development/fuzz_targets.go new file mode 100644 index 00000000..45066c40 --- /dev/null +++ b/helpers/internal/development/fuzz_targets.go @@ -0,0 +1,43 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +// Labels, corpus identities and limits share one inventory so a component-local +// harness cannot silently disappear from the nightly run after a directory move. +type fuzzTarget struct { + label, binary, corpus, maxLength, rssLimit string + // frontend targets load the Haskell shared library beside the executable. + frontend bool + // threaded targets start their own worker threads. Only these gain evidence + // from a ThreadSanitizer campaign; the others execute one thread per input. + threaded bool +} + +func nativeFuzzTargets() []fuzzTarget { + return []fuzzTarget{ + {"//Compiler/Fuzzing:wire_fuzzer", "wire_fuzzer", "wire", "16384", "768", false, false}, + {"//Compiler/Fuzzing:lexer_fuzzer", "lexer_fuzzer", "lexer", "65536", "1024", true, false}, + {"//Compiler/Fuzzing:parser_fuzzer", "parser_fuzzer", "parser", "65536", "1536", true, false}, + {"//Compiler/Fuzzing:source_llvm_fuzzer", "source_llvm_fuzzer", "source", "65536", "4096", true, false}, + // The generator reads at most 33 bytes: 31 expression selectors at depth + // four, then one shape and one trip-count byte. Bound mutations close to + // that semantic input instead of evolving unused tails. + {"//Compiler/Fuzzing:differential_fuzzer", "differential_fuzzer", "differential", "64", "4096", true, false}, + {"//Compiler/Cli/Fuzzing:cli_fuzzer", "cli_fuzzer", "cli", "16384", "768", false, false}, + {"//Compiler/ProjectSystem/Bridge/Fuzzing:project_fuzzer", "project_fuzzer", "project", "65536", "768", false, false}, + {"//Interactive/Fuzzing:repl_fuzzer", "repl_fuzzer", "repl", "64", "4096", true, false}, + {"//Compiler/Runtime/AARC/Fuzzing:ownership_fuzzer", "ownership_fuzzer", "ownership", "64", "768", false, true}, + } +} + +// threadFuzzTargets selects the harnesses whose oracle depends on interleaving. +func threadFuzzTargets() []fuzzTarget { + var targets []fuzzTarget + for _, target := range nativeFuzzTargets() { + if target.threaded { + targets = append(targets, target) + } + } + return targets +} diff --git a/helpers/internal/development/fuzz_thread.go b/helpers/internal/development/fuzz_thread.go new file mode 100644 index 00000000..f673ee0b --- /dev/null +++ b/helpers/internal/development/fuzz_thread.go @@ -0,0 +1,125 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "errors" + "fmt" + "os" + "path/filepath" +) + +// threadFuzzBuildArguments instruments the threaded harnesses for libFuzzer and +// ThreadSanitizer together. AddressSanitizer cannot share a binary with +// ThreadSanitizer, so this is a separate build and a separate campaign. +func threadFuzzBuildArguments(configuration, sanitizerConfiguration, macRuntime string) []string { + arguments := []string{"build", "--config=" + configuration, "--config=" + sanitizerConfiguration} + if macRuntime != "" { + arguments = append(arguments, "--linkopt="+macRuntime) + } + for _, target := range threadFuzzTargets() { + arguments = append(arguments, target.label) + } + return arguments +} + +// runThreadFuzzCampaign runs the interleaving-dependent harnesses under +// ThreadSanitizer. A host without a ThreadSanitizer runtime is an error, never +// a skipped success: the selection fails on Windows, and the runtime probe must +// start cleanly and report its intentional data race before any input runs. +func runThreadFuzzCampaign(repository string, currentHost host, runner commandRunner) error { + duration, err := fuzzDuration(false, os.Getenv("VXS_FUZZ_SECONDS")) + if err != nil { + return err + } + selected, err := selectSanitizer(currentHost, "thread") + if err != nil { + return err + } + configuration, err := fuzzConfiguration(currentHost) + if err != nil { + return err + } + targets := threadFuzzTargets() + if len(targets) == 0 { + return errors.New("no threaded fuzz target is registered") + } + bazel, err := findBazel(runner) + if err != nil { + return err + } + selected.environment, err = sanitizerEnvironment(currentHost, selected, runner) + if err != nil { + return err + } + if err := verifySanitizerRuntime(repository, currentHost, runner, selected); err != nil { + return err + } + temporaryRoot := os.Getenv("RUNNER_TEMP") + if temporaryRoot == "" { + temporaryRoot = os.TempDir() + } + work, err := os.MkdirTemp(temporaryRoot, "vxs-fuzz-") + if err != nil { + return fmt.Errorf("could not create fuzz work directory: %w", err) + } + persistentCorpus := os.Getenv("VXS_FUZZ_CORPUS") + corpusRoot := persistentCorpus + if corpusRoot == "" { + corpusRoot = filepath.Join(work, "corpus") + } + artifacts := filepath.Join(work, "artifacts") + if err := os.Mkdir(artifacts, 0o700); err != nil { + return fmt.Errorf("could not create fuzz artifact directory %q: %w", work, err) + } + for _, target := range targets { + corpus := filepath.Join(corpusRoot, target.corpus) + if err := os.MkdirAll(corpus, 0o700); err != nil { + return fmt.Errorf("could not prepare %s corpus; preserved %q: %w", target.corpus, work, err) + } + if err := syncSeedCorpus(filepath.Join(repository, "Compiler", "Fuzzing", "Corpus", target.corpus), corpus); err != nil { + return fmt.Errorf("could not synchronize the versioned %s seed corpus: %w", target.corpus, err) + } + } + macRuntime := "" + if currentHost.kind == hostMacOS { + macRuntime, err = macOSFuzzerRuntime(os.Getenv("LLVM_ROOT")) + if err != nil { + return fmt.Errorf("could not locate macOS libFuzzer runtime; preserved %q: %w", work, err) + } + } + if err := runner.Run(repository, nil, bazel, threadFuzzBuildArguments(configuration, selected.config, macRuntime)...); err != nil { + return fmt.Errorf("could not build ThreadSanitizer fuzz targets; preserved %q: %w", work, err) + } + campaign := fuzzCampaign{ + repository: repository, + corpusRoot: corpusRoot, + artifacts: artifacts, + work: work, + report: "thread-campaigns.json", + duration: duration, + environment: selected.environment, + sanitizer: selected.config, + executable: currentHost.executable, + } + for _, target := range targets { + if target.frontend { + // The GHC runtime is not ThreadSanitizer-instrumented; a frontend + // target here would report races this campaign cannot attribute. + return fmt.Errorf("threaded fuzz target %s must not depend on the Haskell frontend", target.binary) + } + if err := campaign.run(runner, target, ""); err != nil { + return err + } + } + if os.Getenv("CI") == "true" { + fmt.Printf("ThreadSanitizer campaign logs preserved in %s.\n", work) + return nil + } + if err := removeSuccessfulFuzzWork(temporaryRoot, work); err != nil { + return err + } + fmt.Println("All ThreadSanitizer fuzz targets completed without a reported failure.") + return nil +} diff --git a/helpers/internal/development/fuzz_thread_test.go b/helpers/internal/development/fuzz_thread_test.go new file mode 100644 index 00000000..3a88390b --- /dev/null +++ b/helpers/internal/development/fuzz_thread_test.go @@ -0,0 +1,44 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "strings" + "testing" +) + +func TestThreadFuzzPlanInstrumentsOnlyThreadedTargets(t *testing.T) { + targets := threadFuzzTargets() + if len(targets) != 1 || targets[0].binary != "ownership_fuzzer" || targets[0].frontend { + t.Fatalf("unexpected threaded targets: %+v", targets) + } + for _, example := range []struct{ fuzz, sanitizer, runtime string }{ + {"fuzz-linux", "tsan-linux", ""}, + {"fuzz-macos", "tsan-macos", "/llvm/libclang_rt.fuzzer_osx.a"}, + } { + plan := threadFuzzBuildArguments(example.fuzz, example.sanitizer, example.runtime) + joined := strings.Join(plan, " ") + for _, required := range []string{"--config=" + example.fuzz, "--config=" + example.sanitizer, "//Compiler/Runtime/AARC/Fuzzing:ownership_fuzzer"} { + if !strings.Contains(joined, required) { + t.Fatalf("thread plan lacks %s: %v", required, plan) + } + } + // ASan and TSan runtimes cannot coexist in one executable. + if strings.Contains(joined, "asan") || strings.Contains(joined, "_smoke") { + t.Fatalf("thread plan mixes incompatible instrumentation: %v", plan) + } + if (example.runtime != "") != strings.Contains(joined, "--linkopt=") { + t.Fatalf("libFuzzer runtime link option mismatch: %v", plan) + } + } +} + +func TestThreadFuzzIsRejectedWhereNoRuntimeExists(t *testing.T) { + // Windows has no Clang ThreadSanitizer runtime. The campaign must fail + // before building or executing anything instead of reporting success. + err := runThreadFuzzCampaign(t.TempDir(), host{kind: hostWindows, executable: ".exe"}, nil) + if err == nil || !strings.Contains(err.Error(), "does not support Windows ThreadSanitizer") { + t.Fatalf("unsupported host was not rejected explicitly: %v", err) + } +} diff --git a/helpers/internal/development/process.go b/helpers/internal/development/process.go index 3f737c39..a5820581 100644 --- a/helpers/internal/development/process.go +++ b/helpers/internal/development/process.go @@ -4,11 +4,16 @@ package development import ( + "context" + "fmt" "io" "os" "os/exec" + "path/filepath" "runtime" + "strconv" "strings" + "time" ) type commandRunner interface { @@ -25,24 +30,82 @@ type systemRunner struct { } func (runner systemRunner) Run(directory string, environment []string, name string, arguments ...string) error { - command := exec.Command(name, arguments...) + command, finish := watchedCommand(name, arguments) + defer finish(nil) command.Dir = directory command.Env = mergedEnvironment(environment) command.Stdin = os.Stdin command.Stdout = runner.stdout command.Stderr = runner.stderr - return command.Run() + return finish(command.Run()) } // RunWithInput drives programs with deterministic stdin while keeping captured // output available to the caller for smoke-test assertions. func (runner systemRunner) RunWithInput(directory string, environment []string, input string, name string, arguments ...string) (string, error) { - command := exec.Command(name, arguments...) + command, finish := watchedCommand(name, arguments) + defer finish(nil) command.Dir = directory command.Env = mergedEnvironment(environment) command.Stdin = strings.NewReader(input) output, err := command.CombinedOutput() - return string(output), err + return string(output), finish(err) +} + +// watchdogGrace bounds how long a cancelled program may hold its output pipes +// open after its process tree was terminated. +const watchdogGrace = 10 * time.Second + +// watchedCommand bounds fuzz and smoke programs. libFuzzer's -timeout covers one +// input, but a deterministic smoke program or an HPC campaign has no in-process +// watchdog, and a miscompiled generated loop never returns. The deadline +// terminates the whole process tree rather than only the direct child, and the +// returned finish function reports expiry as a watchdog failure instead of the +// host's generic kill status. Build tools are not bounded here; they keep their +// CI job timeout. +func watchedCommand(name string, arguments []string) (*exec.Cmd, func(error) error) { + return watchedCommandWithin(fuzzProcessSeconds(name, arguments), name, arguments) +} + +func watchedCommandWithin(seconds int, name string, arguments []string) (*exec.Cmd, func(error) error) { + if seconds == 0 { + return exec.Command(name, arguments...), func(err error) error { return err } + } + ctx, cancel := context.WithTimeout(context.Background(), time.Duration(seconds)*time.Second) + command := exec.CommandContext(ctx, name, arguments...) + prepareProcessTree(command) + command.Cancel = func() error { return terminateProcessTree(command) } + command.WaitDelay = watchdogGrace + return command, func(err error) error { + expired := ctx.Err() == context.DeadlineExceeded + cancel() + if err != nil && expired { + return fmt.Errorf("%s exceeded its %d-second process watchdog and its process tree was terminated: %w", filepath.Base(name), seconds, err) + } + return err + } +} + +func fuzzProcessSeconds(name string, arguments []string) int { + binary := strings.TrimSuffix(filepath.Base(name), ".exe") + if binary == "source_fuzz_smoke" || binary == "wire_fuzz_smoke" { + return 90 + } + if binary == "frontend-fuzz" && len(arguments) >= 2 { + if seconds, err := strconv.Atoi(arguments[1]); err == nil && seconds >= 1 && seconds <= 3600 { + return seconds + 300 // bounded corpus warmup and shutdown allowance + } + } + if strings.HasSuffix(binary, "_fuzzer") { + for _, argument := range arguments { + if value, ok := strings.CutPrefix(argument, "-max_total_time="); ok { + if seconds, err := strconv.Atoi(value); err == nil && seconds >= 1 && seconds <= 3600 { + return seconds + 90 + } + } + } + } + return 0 } // mergedEnvironment replaces inherited keys rather than appending duplicate diff --git a/helpers/internal/development/process_alive_unix_test.go b/helpers/internal/development/process_alive_unix_test.go new file mode 100644 index 00000000..3a25acda --- /dev/null +++ b/helpers/internal/development/process_alive_unix_test.go @@ -0,0 +1,13 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +//go:build !windows + +package development + +import "syscall" + +// processAlive reports whether a process with this identifier still exists. +func processAlive(pid int) bool { + return syscall.Kill(pid, 0) == nil +} diff --git a/helpers/internal/development/process_alive_windows_test.go b/helpers/internal/development/process_alive_windows_test.go new file mode 100644 index 00000000..d0c1fd64 --- /dev/null +++ b/helpers/internal/development/process_alive_windows_test.go @@ -0,0 +1,19 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +//go:build windows + +package development + +import ( + "os/exec" + "strconv" + "strings" +) + +// processAlive reports whether a process with this identifier still exists. +// os.FindProcess always succeeds on Windows, so query the process table. +func processAlive(pid int) bool { + output, err := exec.Command("tasklist", "/NH", "/FI", "PID eq "+strconv.Itoa(pid)).Output() + return err == nil && strings.Contains(string(output), " "+strconv.Itoa(pid)+" ") +} diff --git a/helpers/internal/development/process_test.go b/helpers/internal/development/process_test.go new file mode 100644 index 00000000..4b32bf09 --- /dev/null +++ b/helpers/internal/development/process_test.go @@ -0,0 +1,90 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "os" + "os/exec" + "path/filepath" + "strconv" + "strings" + "testing" + "time" +) + +const watchdogRole = "VXS_WATCHDOG_TEST_ROLE" + +// TestMain lets the test binary stand in for a hung fuzz program: the parent +// role starts a child that never exits and then blocks, like a smoke program +// stuck in a non-terminating generated loop with a helper process below it. +func TestMain(m *testing.M) { + switch os.Getenv(watchdogRole) { + case "parent": + child := exec.Command(os.Args[0]) + child.Env = append(os.Environ(), watchdogRole+"=child") + child.Stdout, child.Stderr = os.Stdout, os.Stderr + if err := child.Start(); err != nil { + os.Exit(3) + } + if err := os.WriteFile(os.Getenv("VXS_WATCHDOG_TEST_PID"), []byte(strconv.Itoa(child.Process.Pid)), 0o600); err != nil { + os.Exit(4) + } + time.Sleep(10 * time.Minute) + os.Exit(5) + case "child": + time.Sleep(10 * time.Minute) + os.Exit(6) + } + os.Exit(m.Run()) +} + +func TestWatchdogTerminatesTheWholeProcessTree(t *testing.T) { + pidFile := filepath.Join(t.TempDir(), "child.pid") + t.Setenv(watchdogRole, "parent") + t.Setenv("VXS_WATCHDOG_TEST_PID", pidFile) + command, finish := watchedCommandWithin(3, os.Args[0], nil) + started := time.Now() + output, runErr := command.CombinedOutput() + err := finish(runErr) + elapsed := time.Since(started) + if err == nil || !strings.Contains(err.Error(), "exceeded its 3-second process watchdog") { + t.Fatalf("hung program was not reported as a watchdog failure: %v\n%s", err, output) + } + // The deadline plus the pipe grace period bounds the wait; without tree + // termination the grandchild keeps the output pipe open for minutes. + if elapsed > 3*time.Second+watchdogGrace+5*time.Second { + t.Fatalf("watchdog returned only after %s", elapsed) + } + text, readErr := os.ReadFile(pidFile) + if readErr != nil { + t.Fatalf("parent never started its child: %v", readErr) + } + pid, convErr := strconv.Atoi(string(text)) + if convErr != nil { + t.Fatal(convErr) + } + deadline := time.Now().Add(10 * time.Second) + for processAlive(pid) { + if time.Now().After(deadline) { + t.Fatalf("descendant %d survived the watchdog", pid) + } + time.Sleep(100 * time.Millisecond) + } +} + +func TestWatchdogLeavesCompletedAndUnboundedCommandsAlone(t *testing.T) { + t.Setenv(watchdogRole, "") + for _, seconds := range []int{0, 60} { + command, finish := watchedCommandWithin(seconds, os.Args[0], []string{"-test.run=^$"}) + if err := finish(command.Run()); err != nil { + t.Fatalf("completed command reported a failure under a %d-second bound: %v", seconds, err) + } + } + // An ordinary failure inside the budget keeps its own error text. + command, finish := watchedCommandWithin(60, os.Args[0], []string{"-test.unknown-flag"}) + err := finish(command.Run()) + if err == nil || strings.Contains(err.Error(), "watchdog") { + t.Fatalf("ordinary failure was misreported: %v", err) + } +} diff --git a/helpers/internal/development/process_unix.go b/helpers/internal/development/process_unix.go new file mode 100644 index 00000000..ce2651f6 --- /dev/null +++ b/helpers/internal/development/process_unix.go @@ -0,0 +1,32 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +//go:build !windows + +package development + +import ( + "errors" + "os" + "os/exec" + "syscall" +) + +// prepareProcessTree places the child in its own process group so descendants +// that it spawns can be signalled together with it. +func prepareProcessTree(command *exec.Cmd) { + command.SysProcAttr = &syscall.SysProcAttr{Setpgid: true} +} + +// terminateProcessTree kills the child's whole process group. A group that has +// already exited is reported as os.ErrProcessDone, which exec treats as benign. +func terminateProcessTree(command *exec.Cmd) error { + if command.Process == nil { + return os.ErrProcessDone + } + err := syscall.Kill(-command.Process.Pid, syscall.SIGKILL) + if errors.Is(err, syscall.ESRCH) { + return os.ErrProcessDone + } + return err +} diff --git a/helpers/internal/development/process_windows.go b/helpers/internal/development/process_windows.go new file mode 100644 index 00000000..1510cef7 --- /dev/null +++ b/helpers/internal/development/process_windows.go @@ -0,0 +1,36 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +//go:build windows + +package development + +import ( + "os" + "os/exec" + "strconv" + "syscall" +) + +// prepareProcessTree detaches the child from the helper's console control +// group so terminating it cannot deliver a console event to the helper itself. +func prepareProcessTree(command *exec.Cmd) { + command.SysProcAttr = &syscall.SysProcAttr{CreationFlags: syscall.CREATE_NEW_PROCESS_GROUP} +} + +// terminateProcessTree ends the child and every descendant. Process.Kill alone +// leaves grandchildren running on Windows, where they keep the inherited output +// pipes open; taskkill /T walks the parent links instead. The direct child is +// killed as well in case taskkill itself is unavailable. +func terminateProcessTree(command *exec.Cmd) error { + if command.Process == nil { + return os.ErrProcessDone + } + tree := exec.Command("taskkill", "/T", "/F", "/PID", strconv.Itoa(command.Process.Pid)) + treeErr := tree.Run() + killErr := command.Process.Kill() + if treeErr == nil || killErr == nil { + return nil + } + return killErr +} diff --git a/justfile b/justfile index 896fd6b2..5c51c5e0 100644 --- a/justfile +++ b/justfile @@ -37,6 +37,10 @@ benchmark: fuzz: go run ./helpers/cmd/develop fuzz +# Threaded fuzz targets under ThreadSanitizer; macOS and native Linux only. +fuzz-thread: + go run ./helpers/cmd/develop fuzz-thread + fuzz-stress: go run ./helpers/cmd/develop fuzz-stress From f5a2777480e57145a999a5f19131d6cd3969b903 Mon Sep 17 00:00:00 2001 From: Leitwolf11 Date: Thu, 1 Oct 2026 23:17:56 +0300 Subject: [PATCH 3/6] Keep fuzz campaigns portable and cut rebuild time without reducing evidence Fix two failures the first hosted run of the new fuzz inventory exposed: - OwnershipFuzzer used std::jthread, which Apple libc++ does not provide. It now uses std::thread and joins every worker before the destructor oracle runs. - The process-tree watchdog test reported a surviving descendant inside the Fedora container. The watchdog had killed it; the orphan stayed a zombie because the container's first process never reaped it, and signal 0 still succeeds for a zombie. The Unix liveness probe now treats a zombie or dead procfs state as terminated. Campaign scheduling: targets and HPC stages run through one bounded scheduler. Concurrency is not free evidence: on a two-core, four-thread host two concurrent targets executed 40 to 90 percent fewer inputs each, so the default is one job per four logical processors, at most four, which keeps such hosts and four-vCPU runners sequential. VXS_FUZZ_JOBS overrides it, targets with a 4096 MiB RSS limit never overlap, and the ThreadSanitizer campaign always runs one target at a time. Budgets, limits, timeouts and watchdogs are identical at every job count, the report keeps inventory order, and the earliest failing target is reported regardless of scheduling. Build time: outside CI the helper adds a persistent content-addressed Bazel disk cache, limited to 8 GiB, to its build invocations. The plain, sanitizer and fuzz configurations share one output tree, and every switch previously recompiled all owned translation units. Measured on the same sources, a bounded campaign dropped from 910 to 436 seconds with unchanged or higher executed-input counts. VXS_BAZEL_DISK_CACHE selects another absolute directory or disables the cache. The fuzzing guide documents the job policy, the measurement behind it and the cache. --- .../Runtime/AARC/Fuzzing/OwnershipFuzzer.cpp | 8 +- Documents/FUZZING.md | 22 +++ helpers/internal/development/bazel_cache.go | 48 +++++++ .../internal/development/bazel_cache_test.go | 51 +++++++ helpers/internal/development/build.go | 4 +- helpers/internal/development/fuzz.go | 70 +++++++--- helpers/internal/development/fuzz_haskell.go | 101 ++++++++------ helpers/internal/development/fuzz_parallel.go | 108 +++++++++++++++ .../development/fuzz_parallel_test.go | 129 ++++++++++++++++++ helpers/internal/development/fuzz_thread.go | 10 +- .../development/process_alive_unix_test.go | 28 +++- helpers/internal/development/sanitizer.go | 2 +- 12 files changed, 505 insertions(+), 76 deletions(-) create mode 100644 helpers/internal/development/bazel_cache.go create mode 100644 helpers/internal/development/bazel_cache_test.go create mode 100644 helpers/internal/development/fuzz_parallel.go create mode 100644 helpers/internal/development/fuzz_parallel_test.go diff --git a/Compiler/Runtime/AARC/Fuzzing/OwnershipFuzzer.cpp b/Compiler/Runtime/AARC/Fuzzing/OwnershipFuzzer.cpp index ac2b729a..efbdf96c 100644 --- a/Compiler/Runtime/AARC/Fuzzing/OwnershipFuzzer.cpp +++ b/Compiler/Runtime/AARC/Fuzzing/OwnershipFuzzer.cpp @@ -52,7 +52,9 @@ LLVMFuzzerTestOneInput(const std::uint8_t *data, std::size_t size) = size == 0U ? 2U : 1U + static_cast(data[0] % 4U); const auto iterations = size < 2U ? 8U : 1U + static_cast(data[1] % 64U); - std::vector workers; + // std::jthread is unavailable in Apple libc++; join explicitly below. + std::vector workers; + workers.reserve(workerCount); for (unsigned worker = 0U; worker < workerCount; ++worker) { const auto localWeak = Aarc::CopyWeak(weak); @@ -84,7 +86,9 @@ LLVMFuzzerTestOneInput(const std::uint8_t *data, std::size_t size) } start.store(true, std::memory_order_release); Aarc::ReleaseStrong(payload); - workers.clear(); // jthread joins before the independent destructor oracle. + // Every worker finishes before the independent destructor oracle runs. + for (auto &worker : workers) + worker.join(); if (corrupt.load() || destructions.load() != 1U || Aarc::LockWeak(weak) != nullptr || Aarc::LoadUnowned(unowned) != nullptr) diff --git a/Documents/FUZZING.md b/Documents/FUZZING.md index b8fc37c2..81d6075f 100644 --- a/Documents/FUZZING.md +++ b/Documents/FUZZING.md @@ -79,6 +79,28 @@ target; `fuzz-stress` defaults to 900. Set `VXS_FUZZ_SECONDS` to an integer from 1 through 3600 to override either duration. CI uses 90 seconds per target for bounded campaigns and 900 for scheduled stress campaigns. +Targets are independent processes with their own corpus, artifact directory +and report entry, so the helper can run several at once. Concurrency is not +free evidence: a campaign is worth the inputs it executes inside its time +budget, and on a two-core, four-thread host two concurrent targets executed +40 to 90 percent fewer inputs each. The default is therefore one job per four +logical processors, at most four, which is one target at a time on such a host +and on four-vCPU CI runners. Set `VXS_FUZZ_JOBS` to an integer from 1 through +64 on a host with spare cores. Two targets with a 4096 MiB RSS limit never +overlap, the three HPC stages follow the same setting, and the ThreadSanitizer +campaign always runs one target at a time because its targets start their own +threads. Time budgets, RSS limits, per-input timeouts and watchdogs are +identical at every job count. + +Most local wall-clock time is compilation, not fuzzing. The plain, sanitizer +and fuzz configurations share one Bazel output tree, and each switch would +otherwise recompile every owned translation unit. Outside CI the helper adds a +persistent content-addressed Bazel disk cache under the user cache directory +(`visual-xsharp/bazel-disk-cache`, limited to 8 GiB). It is keyed by each +action's full command line and inputs, so no instrumentation or check changes. +`VXS_BAZEL_DISK_CACHE` selects another absolute directory, or `off` for a cold +measurement. + Each campaign has a 30-second per-input timeout and a finite input length: | Target | Maximum input bytes | RSS limit (MiB) | diff --git a/helpers/internal/development/bazel_cache.go b/helpers/internal/development/bazel_cache.go new file mode 100644 index 00000000..d5c052ff --- /dev/null +++ b/helpers/internal/development/bazel_cache.go @@ -0,0 +1,48 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "os" + "path/filepath" +) + +// bazelDiskCacheLimit bounds the local cache; Bazel evicts the least recently +// used entries beyond it while idle. +const bazelDiskCacheLimit = "8G" + +// cachedBuild adds a persistent, content-addressed disk cache to a Bazel build. +// One output tree serves the plain, sanitizer and fuzz configurations in turn, +// and Bazel's in-tree action cache remembers only the most recent action per +// output, so every configuration switch otherwise recompiles all owned +// translation units even when no source changed. The disk cache is keyed by +// each action's complete command line and inputs: a sanitizer or fuzz object +// can never be served to another configuration, and no instrumentation, +// check or time budget is altered. +// +// CI keeps its own cache through the workflow's Bazel setup, and +// VXS_BAZEL_DISK_CACHE=off disables this one for a cold local measurement. +func cachedBuild(arguments []string) []string { + directory, enabled := bazelDiskCacheDirectory(os.Getenv("CI"), os.Getenv("VXS_BAZEL_DISK_CACHE"), os.UserCacheDir) + if !enabled || len(arguments) == 0 || arguments[0] != "build" { + return arguments + } + cached := make([]string, 0, len(arguments)+2) + cached = append(cached, "build", "--disk_cache="+directory, "--experimental_disk_cache_gc_max_size="+bazelDiskCacheLimit) + return append(cached, arguments[1:]...) +} + +func bazelDiskCacheDirectory(ci, configured string, userCache func() (string, error)) (string, bool) { + if ci == "true" || configured == "off" { + return "", false + } + if configured != "" { + return configured, filepath.IsAbs(configured) + } + root, err := userCache() + if err != nil || root == "" { + return "", false + } + return filepath.Join(root, "visual-xsharp", "bazel-disk-cache"), true +} diff --git a/helpers/internal/development/bazel_cache_test.go b/helpers/internal/development/bazel_cache_test.go new file mode 100644 index 00000000..e97030eb --- /dev/null +++ b/helpers/internal/development/bazel_cache_test.go @@ -0,0 +1,51 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "errors" + "path/filepath" + "testing" +) + +func TestBazelDiskCacheIsLocalOptionalAndAbsolute(t *testing.T) { + root := t.TempDir() + cache := func() (string, error) { return root, nil } + if directory, enabled := bazelDiskCacheDirectory("", "", cache); !enabled || directory != filepath.Join(root, "visual-xsharp", "bazel-disk-cache") { + t.Fatalf("default cache: %q, %v", directory, enabled) + } + // CI owns its cache through the workflow; a developer may opt out. + for _, example := range [][2]string{{"true", ""}, {"", "off"}, {"true", root}} { + if directory, enabled := bazelDiskCacheDirectory(example[0], example[1], cache); enabled { + t.Fatalf("cache enabled for CI=%q setting=%q: %q", example[0], example[1], directory) + } + } + if directory, enabled := bazelDiskCacheDirectory("", root, cache); !enabled || directory != root { + t.Fatalf("explicit cache: %q, %v", directory, enabled) + } + // A relative path would resolve inside Bazel's own working directory. + if _, enabled := bazelDiskCacheDirectory("", "relative/cache", cache); enabled { + t.Fatal("relative cache directory accepted") + } + if _, enabled := bazelDiskCacheDirectory("", "", func() (string, error) { return "", errors.New("no cache directory") }); enabled { + t.Fatal("cache enabled without a user cache directory") + } +} + +func TestCachedBuildOnlyExtendsBuildCommands(t *testing.T) { + t.Setenv("CI", "") + t.Setenv("VXS_BAZEL_DISK_CACHE", t.TempDir()) + build := cachedBuild([]string{"build", "--config=asan-linux", "//Compiler/Cli:vxs"}) + if len(build) != 5 || build[0] != "build" || build[3] != "--config=asan-linux" || build[4] != "//Compiler/Cli:vxs" { + t.Fatalf("build arguments were reordered or dropped: %v", build) + } + clean := []string{"clean", "--expunge"} + if got := cachedBuild(clean); len(got) != 2 { + t.Fatalf("non-build command changed: %v", got) + } + t.Setenv("VXS_BAZEL_DISK_CACHE", "off") + if got := cachedBuild([]string{"build", "//Compiler/Cli:vxs"}); len(got) != 2 { + t.Fatalf("disabled cache still added options: %v", got) + } +} diff --git a/helpers/internal/development/build.go b/helpers/internal/development/build.go index c89c7206..2750aec6 100644 --- a/helpers/internal/development/build.go +++ b/helpers/internal/development/build.go @@ -79,7 +79,7 @@ func buildTargets(repository string, runner commandRunner, config string, extra arguments = append(arguments, nativeTargets...) arguments = append(arguments, extra...) fmt.Printf("Building compiler and %d native suites...\n", len(nativeTargets)) - if err := runner.Run(repository, nil, bazel, arguments...); err != nil { + if err := runner.Run(repository, nil, bazel, cachedBuild(arguments)...); err != nil { return fmt.Errorf("Bazel build failed: %w", err) } if err := stageFrontendForBuildOutputs(repository, frontendLibrary); err != nil { @@ -184,7 +184,7 @@ func runBenchmarks(repository string, currentHost host, runner commandRunner, ba arguments := append([]string{"build", "-c", "opt"}, nativeBenchmarkTargets...) arguments = append(arguments, bazelArguments...) fmt.Printf("Building %d native benchmark programs...\n", len(nativeBenchmarkTargets)) - if err := runner.Run(repository, nil, bazel, arguments...); err != nil { + if err := runner.Run(repository, nil, bazel, cachedBuild(arguments)...); err != nil { return fmt.Errorf("native benchmark build failed: %w", err) } for index, program := range nativeBenchmarkPrograms { diff --git a/helpers/internal/development/fuzz.go b/helpers/internal/development/fuzz.go index 17206f6f..3c2cc46a 100644 --- a/helpers/internal/development/fuzz.go +++ b/helpers/internal/development/fuzz.go @@ -13,6 +13,7 @@ import ( "path/filepath" "strconv" "strings" + "sync" ) func fuzzConfiguration(currentHost host) (string, error) { @@ -150,7 +151,7 @@ func runFuzzCampaign(repository string, currentHost host, runner commandRunner, macRuntime = runtime } smokeArguments, campaignArguments := fuzzBuildArguments(configuration, sanitizerConfiguration, macRuntime) - if err := runner.Run(repository, nil, bazel, smokeArguments...); err != nil { + if err := runner.Run(repository, nil, bazel, cachedBuild(smokeArguments)...); err != nil { return fmt.Errorf("could not build standalone fuzz smoke targets; preserved %q: %w", work, err) } if err := runner.Run(repository, selectedEnvironment, smoke, "-Write-Corpus", wireGenerated); err != nil { @@ -168,7 +169,7 @@ func runFuzzCampaign(repository string, currentHost host, runner commandRunner, } // Run smoke tests before changing Bazel's instrumentation configuration and // staging the campaign binaries into the same host output tree. - if err := runner.Run(repository, nil, bazel, campaignArguments...); err != nil { + if err := runner.Run(repository, nil, bazel, cachedBuild(campaignArguments)...); err != nil { return fmt.Errorf("could not build instrumented fuzz targets; preserved %q: %w", work, err) } campaign := fuzzCampaign{ @@ -182,12 +183,14 @@ func runFuzzCampaign(repository string, currentHost host, runner commandRunner, sanitizer: sanitizerConfiguration, executable: currentHost.executable, } - for _, target := range nativeFuzzTargets() { - if err := campaign.run(runner, target, frontendLibrary); err != nil { - return err - } + jobs, err := hostFuzzJobs(os.Getenv("VXS_FUZZ_JOBS")) + if err != nil { + return err } - if err := runHaskellFuzz(repository, corpusRoot, artifacts, duration, runner); err != nil { + if err := campaign.runAll(runner, nativeFuzzTargets(), frontendLibrary, jobs); err != nil { + return err + } + if err := runHaskellFuzz(repository, corpusRoot, artifacts, duration, jobs, runner); err != nil { return fmt.Errorf("Haskell feedback campaign failed; preserved %q: %w", work, err) } if os.Getenv("CI") == "true" { @@ -217,21 +220,42 @@ type fuzzCampaign struct { duration int environment []string sanitizer, executable string - records []map[string]any + // records holds one slot per target in inventory order; guard serializes + // report updates and console output of concurrently finishing targets. + records []map[string]any + guard sync.Mutex +} + +// runAll stages every frontend library first, because several targets share +// one output directory, and then runs the targets with bounded concurrency. +// Each target keeps its own corpus, artifact directory, time budget, RSS +// limit, watchdog and report checks exactly as in a sequential run. +func (campaign *fuzzCampaign) runAll(runner commandRunner, targets []fuzzTarget, frontendLibrary string, jobs int) error { + campaign.records = make([]map[string]any, len(targets)) + tasks := make([]fuzzTask, 0, len(targets)) + for index, target := range targets { + if target.frontend { + if err := copyFile(frontendLibrary, filepath.Join(filepath.Dir(campaign.program(target)), filepath.Base(frontendLibrary)), 0o755); err != nil { + return fmt.Errorf("could not stage frontend for %s; preserved %q: %w", target.binary, campaign.work, err) + } + } + tasks = append(tasks, fuzzTask{heavy: isHeavyFuzzTarget(target), run: func() error { return campaign.run(runner, index, target) }}) + } + fmt.Printf("Running %d fuzz targets with up to %d concurrent processes.\n", len(targets), jobs) + return runFuzzTasks(jobs, tasks) +} + +func (campaign *fuzzCampaign) program(target fuzzTarget) string { + packagePath, _, _ := strings.Cut(strings.TrimPrefix(target.label, "//"), ":") + return filepath.Join(campaign.repository, "bazel-bin", filepath.FromSlash(packagePath), target.binary+campaign.executable) } // run executes one instrumented target against its persistent corpus. The // report is rewritten after every target so a later failure still leaves the // earlier measurements, and a process that exits zero without libFuzzer's final // counters is a failure rather than an unexplained success. -func (campaign *fuzzCampaign) run(runner commandRunner, target fuzzTarget, frontendLibrary string) error { - packagePath, _, _ := strings.Cut(strings.TrimPrefix(target.label, "//"), ":") - fuzzer := filepath.Join(campaign.repository, "bazel-bin", filepath.FromSlash(packagePath), target.binary+campaign.executable) - if target.frontend { - if err := copyFile(frontendLibrary, filepath.Join(filepath.Dir(fuzzer), filepath.Base(frontendLibrary)), 0o755); err != nil { - return fmt.Errorf("could not stage frontend for %s; preserved %q: %w", target.binary, campaign.work, err) - } - } +func (campaign *fuzzCampaign) run(runner commandRunner, index int, target fuzzTarget) error { + fuzzer := campaign.program(target) campaignArtifacts := filepath.Join(campaign.artifacts, target.binary) if err := os.MkdirAll(campaignArtifacts, 0o700); err != nil { return fmt.Errorf("could not create %s artifact directory: %w", target.binary, err) @@ -251,13 +275,21 @@ func (campaign *fuzzCampaign) run(runner commandRunner, target fuzzTarget, front if err := os.WriteFile(filepath.Join(campaign.artifacts, target.binary+".log"), []byte(output), 0o600); err != nil { return fmt.Errorf("could not preserve campaign log: %w", err) } - fmt.Print(output) statistics, statisticsErr := parseFuzzStatistics(output) if runErr == nil && statisticsErr != nil { runErr = statisticsErr } - campaign.records = append(campaign.records, map[string]any{"target": target.binary, "seconds": campaign.duration, "rss_limit_mb": target.rssLimit, "sanitizer": campaign.sanitizer, "native_coverage": true, "haskell_native_coverage": false, "statistics": statistics, "success": runErr == nil}) - report, err := json.MarshalIndent(campaign.records, "", " ") + campaign.guard.Lock() + defer campaign.guard.Unlock() + fmt.Print(output) + campaign.records[index] = map[string]any{"target": target.binary, "seconds": campaign.duration, "rss_limit_mb": target.rssLimit, "sanitizer": campaign.sanitizer, "native_coverage": true, "haskell_native_coverage": false, "statistics": statistics, "success": runErr == nil} + finished := make([]map[string]any, 0, len(campaign.records)) + for _, record := range campaign.records { + if record != nil { + finished = append(finished, record) + } + } + report, err := json.MarshalIndent(finished, "", " ") if err != nil { return err } diff --git a/helpers/internal/development/fuzz_haskell.go b/helpers/internal/development/fuzz_haskell.go index b2ea1f2b..fbba00db 100644 --- a/helpers/internal/development/fuzz_haskell.go +++ b/helpers/internal/development/fuzz_haskell.go @@ -10,12 +10,13 @@ import ( "path/filepath" "strconv" "strings" + "sync" ) // The native C ABI wrapper cannot feed back GHC branches. This executable is // rebuilt in an isolated HPC tree and retains inputs that hit new production // ticks, rather than pretending native callback coverage represents Haskell. -func runHaskellFuzz(repository, corpusRoot, artifacts string, duration int, runner commandRunner) error { +func runHaskellFuzz(repository, corpusRoot, artifacts string, duration, jobs int, runner commandRunner) error { directory := filepath.Join(repository, "Compiler") options := []string{"--enable-coverage", "--disable-tests", "--builddir=dist-fuzz-coverage"} build := append([]string{"build", "exe:frontend-fuzz"}, options...) @@ -27,53 +28,63 @@ func runHaskellFuzz(repository, corpusRoot, artifacts string, duration int, runn if err != nil || !filepath.IsAbs(binary) || strings.ContainsAny(binary, "\r\n") { return fmt.Errorf("could not resolve HPC frontend executable: %q (%v)", binary, err) } - for _, stage := range []string{"lexer", "parser", "source"} { - corpus := filepath.Join(corpusRoot, "haskell-"+stage) - resultPath := filepath.Join(artifacts, "haskell-"+stage) - for _, path := range []string{corpus, resultPath} { - if err := os.MkdirAll(path, 0o700); err != nil { + // Stages share only the read-only executable: each has its own corpus, + // tick file, report and artifact directory, so they may run together. + var console sync.Mutex + stages := []string{"lexer", "parser", "source"} + tasks := make([]fuzzTask, 0, len(stages)) + for _, stage := range stages { + tasks = append(tasks, fuzzTask{run: func() error { + corpus := filepath.Join(corpusRoot, "haskell-"+stage) + resultPath := filepath.Join(artifacts, "haskell-"+stage) + for _, path := range []string{corpus, resultPath} { + if err := os.MkdirAll(path, 0o700); err != nil { + return err + } + } + if err := syncSeedCorpus(filepath.Join(repository, "Compiler", "Fuzzing", "Corpus", stage), corpus); err != nil { return err } - } - if err := syncSeedCorpus(filepath.Join(repository, "Compiler", "Fuzzing", "Corpus", stage), corpus); err != nil { - return err - } - // An HPC executable loads an existing tick file at startup and aborts - // when its module hashes belong to an earlier build. Give every stage a - // fresh file beside its report instead of the default one in the - // working directory, which would also leave output in the source tree. - tickFile := filepath.Join(resultPath, "frontend-fuzz.tix") - if err := os.Remove(tickFile); err != nil && !os.IsNotExist(err) { - return fmt.Errorf("could not reset the HPC tick file: %w", err) - } - // The engine writes its report to this file only after a completed - // campaign. Remove an earlier one so a crashed run cannot inherit it. - reportFile := filepath.Join(resultPath, "campaign.txt") - if err := os.Remove(reportFile); err != nil && !os.IsNotExist(err) { - return fmt.Errorf("could not reset the HPC campaign report: %w", err) - } - output, runErr := runner.RunWithInput(directory, []string{"HPCTIXFILE=" + tickFile}, "", binary, stage, strconv.Itoa(duration), corpus, resultPath, "12345") - fmt.Print(output) - if err := os.WriteFile(filepath.Join(resultPath, "output.log"), []byte(output), 0o600); err != nil { - return err - } - stats, statsErr := readHaskellFuzzReport(reportFile, output, stage) - record := map[string]any{"stage": stage, "seconds": duration, "coverage_engine": "ghc-hpc", "asan_instrumented": false, "heap_limit_mib": 512, "statistics": stats, "success": runErr == nil && statsErr == nil} - report, err := json.MarshalIndent(record, "", " ") - if err != nil { - return err - } - if err := os.WriteFile(filepath.Join(resultPath, "campaign.json"), report, 0o600); err != nil { - return err - } - if runErr != nil { - return fmt.Errorf("HPC %s campaign failed; artifacts in %q: %w", stage, resultPath, runErr) - } - if statsErr != nil { - return statsErr - } + // An HPC executable loads an existing tick file at startup and aborts + // when its module hashes belong to an earlier build. Give every stage a + // fresh file beside its report instead of the default one in the + // working directory, which would also leave output in the source tree. + tickFile := filepath.Join(resultPath, "frontend-fuzz.tix") + if err := os.Remove(tickFile); err != nil && !os.IsNotExist(err) { + return fmt.Errorf("could not reset the HPC tick file: %w", err) + } + // The engine writes its report to this file only after a completed + // campaign. Remove an earlier one so a crashed run cannot inherit it. + reportFile := filepath.Join(resultPath, "campaign.txt") + if err := os.Remove(reportFile); err != nil && !os.IsNotExist(err) { + return fmt.Errorf("could not reset the HPC campaign report: %w", err) + } + output, runErr := runner.RunWithInput(directory, []string{"HPCTIXFILE=" + tickFile}, "", binary, stage, strconv.Itoa(duration), corpus, resultPath, "12345") + console.Lock() + fmt.Print(output) + console.Unlock() + if err := os.WriteFile(filepath.Join(resultPath, "output.log"), []byte(output), 0o600); err != nil { + return err + } + stats, statsErr := readHaskellFuzzReport(reportFile, output, stage) + record := map[string]any{"stage": stage, "seconds": duration, "coverage_engine": "ghc-hpc", "asan_instrumented": false, "heap_limit_mib": 512, "statistics": stats, "success": runErr == nil && statsErr == nil} + report, err := json.MarshalIndent(record, "", " ") + if err != nil { + return err + } + if err := os.WriteFile(filepath.Join(resultPath, "campaign.json"), report, 0o600); err != nil { + return err + } + if runErr != nil { + return fmt.Errorf("HPC %s campaign failed; artifacts in %q: %w", stage, resultPath, runErr) + } + if statsErr != nil { + return statsErr + } + return nil + }}) } - return nil + return runFuzzTasks(jobs, tasks) } // readHaskellFuzzReport validates the report file the engine wrote for this diff --git a/helpers/internal/development/fuzz_parallel.go b/helpers/internal/development/fuzz_parallel.go new file mode 100644 index 00000000..f3c65eec --- /dev/null +++ b/helpers/internal/development/fuzz_parallel.go @@ -0,0 +1,108 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "errors" + "runtime" + "strconv" + "sync" +) + +// heavyRSSLimitMiB marks targets whose RSS limit is large enough that two of +// them running together could exhaust a developer machine. Their limits are +// not lowered; they simply never overlap each other. +const heavyRSSLimitMiB = 4096 + +// fuzzTask is one independent campaign process. Tasks share no corpus, +// artifact directory or report entry, so running them together changes only +// the wall-clock time, not the per-target budget or its checks. +type fuzzTask struct { + heavy bool + run func() error +} + +// fuzzJobs selects how many campaign processes may run at once. A campaign's +// evidence is the number of inputs a target executes inside its time budget, +// so concurrency is only free while every process keeps a physical core to +// itself. Measured on a two-core, four-thread host, two concurrent targets +// cut executed inputs by 40 to 90 percent. The default therefore grants one +// job per four logical processors, at most four, which is a single job on +// such a host and on four-vCPU CI runners. VXS_FUZZ_JOBS overrides it for a +// host with spare cores. +func fuzzJobs(configured string, logicalProcessors int) (int, error) { + if configured != "" { + jobs, err := strconv.Atoi(configured) + if err != nil || jobs < 1 || jobs > 64 { + return 0, errors.New("VXS_FUZZ_JOBS must be an integer in [1, 64]") + } + return jobs, nil + } + jobs := logicalProcessors / 4 + if jobs < 1 { + jobs = 1 + } + if jobs > 4 { + jobs = 4 + } + return jobs, nil +} + +func hostFuzzJobs(configured string) (int, error) { + return fuzzJobs(configured, runtime.NumCPU()) +} + +// runFuzzTasks executes every task with at most jobs running concurrently and +// at most one heavy task at a time. All started tasks finish so their logs and +// reports are complete; after the first failure no further task is started. +// The returned error is the failure of the earliest task in inventory order, +// which keeps the reported failure independent of scheduling. +func runFuzzTasks(jobs int, tasks []fuzzTask) error { + if jobs < 1 { + jobs = 1 + } + failures := make([]error, len(tasks)) + slots := make(chan struct{}, jobs) + var heavy sync.Mutex + var state sync.Mutex + failed := false + var group sync.WaitGroup + for index, task := range tasks { + slots <- struct{}{} + state.Lock() + stop := failed + state.Unlock() + if stop { + <-slots + break + } + group.Add(1) + go func() { + defer group.Done() + defer func() { <-slots }() + if task.heavy { + heavy.Lock() + defer heavy.Unlock() + } + if err := task.run(); err != nil { + state.Lock() + failures[index] = err + failed = true + state.Unlock() + } + }() + } + group.Wait() + for _, failure := range failures { + if failure != nil { + return failure + } + } + return nil +} + +func isHeavyFuzzTarget(target fuzzTarget) bool { + limit, err := strconv.Atoi(target.rssLimit) + return err != nil || limit >= heavyRSSLimitMiB +} diff --git a/helpers/internal/development/fuzz_parallel_test.go b/helpers/internal/development/fuzz_parallel_test.go new file mode 100644 index 00000000..04258cae --- /dev/null +++ b/helpers/internal/development/fuzz_parallel_test.go @@ -0,0 +1,129 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "errors" + "sync" + "sync/atomic" + "testing" + "time" +) + +func TestFuzzJobsDefaultKeepsAPhysicalCorePerTarget(t *testing.T) { + for _, example := range []struct{ processors, jobs int }{{1, 1}, {2, 1}, {4, 1}, {8, 2}, {16, 4}, {64, 4}} { + got, err := fuzzJobs("", example.processors) + if err != nil || got != example.jobs { + t.Fatalf("%d processors: %d jobs, %v; want %d", example.processors, got, err, example.jobs) + } + } + if got, err := fuzzJobs("1", 16); err != nil || got != 1 { + t.Fatalf("explicit sequential run rejected: %d, %v", got, err) + } + if got, err := fuzzJobs("6", 2); err != nil || got != 6 { + t.Fatalf("explicit job count rejected: %d, %v", got, err) + } + for _, bad := range []string{"0", "-1", "65", "two", "1.5"} { + if _, err := fuzzJobs(bad, 8); err == nil { + t.Fatalf("accepted invalid VXS_FUZZ_JOBS %q", bad) + } + } +} + +func TestFuzzTasksRespectJobAndHeavyLimits(t *testing.T) { + var running, heavyRunning, peak, heavyPeak atomic.Int32 + raise := func(current *atomic.Int32, highest *atomic.Int32) { + value := current.Add(1) + for { + seen := highest.Load() + if value <= seen || highest.CompareAndSwap(seen, value) { + return + } + } + } + var completed atomic.Int32 + tasks := make([]fuzzTask, 0, 12) + for index := 0; index < 12; index++ { + heavy := index%2 == 0 + tasks = append(tasks, fuzzTask{heavy: heavy, run: func() error { + raise(&running, &peak) + if heavy { + raise(&heavyRunning, &heavyPeak) + } + time.Sleep(20 * time.Millisecond) + if heavy { + heavyRunning.Add(-1) + } + running.Add(-1) + completed.Add(1) + return nil + }}) + } + if err := runFuzzTasks(3, tasks); err != nil { + t.Fatal(err) + } + if completed.Load() != 12 { + t.Fatalf("only %d of 12 tasks ran", completed.Load()) + } + if peak.Load() > 3 || peak.Load() < 2 { + t.Fatalf("concurrency peak %d is outside the requested bound of 3", peak.Load()) + } + if heavyPeak.Load() != 1 { + t.Fatalf("%d memory-heavy targets overlapped", heavyPeak.Load()) + } +} + +func TestFuzzTasksReportTheEarliestFailureAndStopStartingWork(t *testing.T) { + first, second := errors.New("first"), errors.New("second") + var started atomic.Int32 + var release sync.WaitGroup + release.Add(1) + tasks := []fuzzTask{ + {run: func() error { started.Add(1); release.Wait(); return first }}, + {run: func() error { started.Add(1); release.Done(); return second }}, + } + for index := 0; index < 20; index++ { + tasks = append(tasks, fuzzTask{run: func() error { started.Add(1); return nil }}) + } + err := runFuzzTasks(2, tasks) + // The later task fails first in time; the reported error is still the + // earliest one in inventory order. + if !errors.Is(err, first) { + t.Fatalf("reported %v instead of the earliest failure", err) + } + if started.Load() == int32(len(tasks)) { + t.Fatal("every task was started although the campaign had already failed") + } + // A single job is exactly the former sequential behavior. + order := []int{} + sequential := []fuzzTask{} + for index := 0; index < 5; index++ { + sequential = append(sequential, fuzzTask{heavy: index == 2, run: func() error { order = append(order, index); return nil }}) + } + if err := runFuzzTasks(1, sequential); err != nil || len(order) != 5 { + t.Fatalf("sequential run: %v, %v", order, err) + } + for index, value := range order { + if index != value { + t.Fatalf("one job did not preserve inventory order: %v", order) + } + } +} + +func TestOnlyLargeRSSTargetsAreHeavy(t *testing.T) { + heavy := map[string]bool{} + for _, target := range nativeFuzzTargets() { + heavy[target.binary] = isHeavyFuzzTarget(target) + } + for _, binary := range []string{"source_llvm_fuzzer", "differential_fuzzer", "repl_fuzzer"} { + if !heavy[binary] { + t.Fatalf("%s has a 4096 MiB limit but may overlap another heavy target", binary) + } + } + for _, binary := range []string{"wire_fuzzer", "lexer_fuzzer", "parser_fuzzer", "cli_fuzzer", "project_fuzzer", "ownership_fuzzer"} { + if heavy[binary] { + t.Fatalf("%s is needlessly serialized", binary) + } + } +} diff --git a/helpers/internal/development/fuzz_thread.go b/helpers/internal/development/fuzz_thread.go index f673ee0b..3e4eff92 100644 --- a/helpers/internal/development/fuzz_thread.go +++ b/helpers/internal/development/fuzz_thread.go @@ -89,7 +89,7 @@ func runThreadFuzzCampaign(repository string, currentHost host, runner commandRu return fmt.Errorf("could not locate macOS libFuzzer runtime; preserved %q: %w", work, err) } } - if err := runner.Run(repository, nil, bazel, threadFuzzBuildArguments(configuration, selected.config, macRuntime)...); err != nil { + if err := runner.Run(repository, nil, bazel, cachedBuild(threadFuzzBuildArguments(configuration, selected.config, macRuntime))...); err != nil { return fmt.Errorf("could not build ThreadSanitizer fuzz targets; preserved %q: %w", work, err) } campaign := fuzzCampaign{ @@ -109,9 +109,11 @@ func runThreadFuzzCampaign(repository string, currentHost host, runner commandRu // target here would report races this campaign cannot attribute. return fmt.Errorf("threaded fuzz target %s must not depend on the Haskell frontend", target.binary) } - if err := campaign.run(runner, target, ""); err != nil { - return err - } + } + // Threaded targets already occupy several cores each and their findings + // depend on scheduling, so they run one at a time. + if err := campaign.runAll(runner, targets, "", 1); err != nil { + return err } if os.Getenv("CI") == "true" { fmt.Printf("ThreadSanitizer campaign logs preserved in %s.\n", work) diff --git a/helpers/internal/development/process_alive_unix_test.go b/helpers/internal/development/process_alive_unix_test.go index 3a25acda..596333a8 100644 --- a/helpers/internal/development/process_alive_unix_test.go +++ b/helpers/internal/development/process_alive_unix_test.go @@ -5,9 +5,31 @@ package development -import "syscall" +import ( + "os" + "strconv" + "strings" + "syscall" +) -// processAlive reports whether a process with this identifier still exists. +// processAlive reports whether a process with this identifier still executes. +// A killed orphan stays in the process table as a zombie until its new parent +// reaps it, and a container's first process may never do so; signal 0 still +// succeeds for such an entry. Where procfs is available, a zombie or dead +// state therefore counts as terminated. func processAlive(pid int) bool { - return syscall.Kill(pid, 0) == nil + if syscall.Kill(pid, 0) != nil { + return false + } + status, err := os.ReadFile("/proc/" + strconv.Itoa(pid) + "/stat") + if err != nil { + return true + } + // The command name is parenthesized and may itself contain spaces or + // parentheses; the state letter follows the last closing parenthesis. + _, rest, found := strings.Cut(string(status[strings.LastIndexByte(string(status), ')')+1:]), " ") + if !found || rest == "" { + return true + } + return rest[0] != 'Z' && rest[0] != 'X' } diff --git a/helpers/internal/development/sanitizer.go b/helpers/internal/development/sanitizer.go index b35c1bc8..04bbe0a3 100644 --- a/helpers/internal/development/sanitizer.go +++ b/helpers/internal/development/sanitizer.go @@ -17,7 +17,7 @@ func verifySanitizerRuntime(repository string, currentHost host, runner commandR if err != nil { return err } - if err := runner.Run(repository, nil, bazel, "build", "--config="+selected.config, "//Compiler/Sanitizers:sanitizer_probe"); err != nil { + if err := runner.Run(repository, nil, bazel, cachedBuild([]string{"build", "--config=" + selected.config, "//Compiler/Sanitizers:sanitizer_probe"})...); err != nil { return fmt.Errorf("sanitizer probe build failed: %w", err) } probe := filepath.Join(repository, "bazel-bin", "Compiler", "Sanitizers", "sanitizer_probe"+currentHost.executable) From 18116c8e694ce9a0f8bde93c1c2dbc752f3952c4 Mon Sep 17 00:00:00 2001 From: Leitwolf11 Date: Thu, 1 Oct 2026 23:55:23 +0300 Subject: [PATCH 4/6] Fix two compiler defects found by the hosted source-to-LLVM fuzz campaign Both inputs are valid programs that passed Core verification and were then rejected inside the native pipeline. Module-wide symbol allocation: the Core-to-CorePrep adapter numbered generated temporaries from the highest identity of the function being lowered and lifted closure names from a separate counter. Identities are unique across a module, so a temporary of one function reused another function's identity and CorePrep verification reported one identity with conflicting spellings and types. Temporaries, condition and short-circuit slots, and closure names now come from one counter that starts above every identity in the module and is never reset, as the Haskell lowering does. Reserved LLVM prefix: a function in a namespace named llvm produced a global whose name began with the prefix LLVM reserves for intrinsics, and module verification failed. Such symbols are emitted with a leading dollar sign, which cannot occur in a source identifier; declarations in other per-source objects follow the same rule. Each defect has a failing-first component test, and both crashing inputs join the versioned source corpus. The Core IR and LLVM backend documents state the allocation and naming rules. --- Compiler/Backend/LLVM/Codegen.cpp | 7 + Compiler/Backend/LLVM/Tests/BUILD.bazel | 1 + .../LLVM/Tests/ReservedSymbolTests.cpp | 87 ++++++++ Compiler/Core/CorePrep/Prepare.cpp | 39 ++-- Compiler/Core/Tests/BUILD.bazel | 1 + Compiler/Core/Tests/SymbolAllocationTests.cpp | 197 ++++++++++++++++++ .../source/cross-function-temporaries.seed | 2 + .../source/reserved-llvm-namespace.seed | 1 + Documents/CORE-IR.md | 9 + Documents/LLVM-BACKEND.md | 9 + 10 files changed, 335 insertions(+), 18 deletions(-) create mode 100644 Compiler/Backend/LLVM/Tests/ReservedSymbolTests.cpp create mode 100644 Compiler/Core/Tests/SymbolAllocationTests.cpp create mode 100644 Compiler/Fuzzing/Corpus/source/cross-function-temporaries.seed create mode 100644 Compiler/Fuzzing/Corpus/source/reserved-llvm-namespace.seed diff --git a/Compiler/Backend/LLVM/Codegen.cpp b/Compiler/Backend/LLVM/Codegen.cpp index e06528b0..a53299d4 100644 --- a/Compiler/Backend/LLVM/Codegen.cpp +++ b/Compiler/Backend/LLVM/Codegen.cpp @@ -131,6 +131,13 @@ namespace Visual::XSharp::Backend::LLVM result->append(*spelling); result->append("."); result->append(std::to_string(function.symbol.id)); + // LLVM reserves every global name beginning with `llvm.` for + // intrinsics and rejects a module that defines one, but `llvm` + // is an ordinary namespace name in source. `$` cannot occur in + // a source identifier, so the prefixed name is unambiguous and + // cannot collide with another module's symbol. + if (result->starts_with("llvm.")) + result->insert(0U, 1U, '$'); return result; } diff --git a/Compiler/Backend/LLVM/Tests/BUILD.bazel b/Compiler/Backend/LLVM/Tests/BUILD.bazel index e389d5ec..4853bfcc 100644 --- a/Compiler/Backend/LLVM/Tests/BUILD.bazel +++ b/Compiler/Backend/LLVM/Tests/BUILD.bazel @@ -9,6 +9,7 @@ cc_binary( "LLVMBackendTests.cpp", "JitSessionTests.cpp", "LoopExecutionTests.cpp", + "ReservedSymbolTests.cpp", "ShortCircuitExecutionTests.cpp", ], deps = [ diff --git a/Compiler/Backend/LLVM/Tests/ReservedSymbolTests.cpp b/Compiler/Backend/LLVM/Tests/ReservedSymbolTests.cpp new file mode 100644 index 00000000..03bc6583 --- /dev/null +++ b/Compiler/Backend/LLVM/Tests/ReservedSymbolTests.cpp @@ -0,0 +1,87 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Backend/LLVM.hpp" +#include "Visual/XSharp/Core/IR.hpp" +#include "Visual/XSharp/Core/Wire.hpp" +#include "Visual/XSharp/Pipeline.hpp" + +// LLVM reserves every global name that begins with `llvm.` for intrinsics +// and rejects a module that defines one. `llvm` is an ordinary Visual X# +// namespace name, so the backend must give such functions a symbol outside +// the reserved prefix instead of failing module verification. Found by the +// source-to-LLVM fuzz campaign. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Llvm = Visual::XSharp::Backend::LLVM; + namespace Pipeline = Visual::XSharp::Pipeline; + + [[nodiscard]] auto + Constant(std::vector moduleName, std::int64_t value) + -> Core::Module + { + return { + std::move(moduleName), + { Core::Function{ + { 1U, U"Evaluate" }, + {}, + Core::Type::int64(), + { Core::Statement::Return( + Core::Expression::Constant(value, Core::Type::int64())) }, + } } + }; + } + + [[nodiscard]] auto + Invoke(const Core::Module &module, std::string_view symbol) -> std::int64_t + { + const auto encoded = Core::Wire::Encode(module); + REQUIRE(encoded); + const auto pipeline = Pipeline::ConsumeCore(encoded.bytes); + if (pipeline.llvm_error) + FAIL_CHECK(pipeline.llvm_error->code + << ": " << pipeline.llvm_error->message); + REQUIRE(pipeline); + REQUIRE(pipeline.llvm); + // LLVM quotes a global name that contains `$`. + const auto quoted = "@\"" + std::string(symbol) + "\"("; + const auto plain = "@" + std::string(symbol) + "("; + CHECK((pipeline.llvm->llvm_ir.find(quoted) != std::string::npos + || pipeline.llvm->llvm_ir.find(plain) != std::string::npos)); + + Llvm::JitSession session; + REQUIRE_FALSE(session.AddModule(pipeline.llvm->bitcode, + "reserved-symbol", + symbol, + Core::Type::int64())); + const auto result = session.InvokeScalar(symbol, Core::Type::int64()); + REQUIRE(result); + return std::get(result.value->payload); + } +} // namespace + +TEST_CASE("functions in a namespace named llvm leave the intrinsic prefix", + "[llvm][symbols]") +{ + CHECK(Invoke(Constant({ U"llvm" }, 11), "$llvm.Evaluate.1") == 11); + CHECK(Invoke(Constant({ U"llvm", U"Fuzz" }, 12), "$llvm.Fuzz.Evaluate.1") + == 12); +} + +TEST_CASE("namespaces that merely resemble the intrinsic prefix are unchanged", + "[llvm][symbols]") +{ + CHECK(Invoke(Constant({ U"llvmx" }, 21), "llvmx.Evaluate.1") == 21); + CHECK(Invoke(Constant({ U"Llvm" }, 22), "Llvm.Evaluate.1") == 22); + CHECK(Invoke(Constant({ U"Demo", U"llvm" }, 23), "Demo.llvm.Evaluate.1") + == 23); +} diff --git a/Compiler/Core/CorePrep/Prepare.cpp b/Compiler/Core/CorePrep/Prepare.cpp index d5664ee8..af0078a8 100644 --- a/Compiler/Core/CorePrep/Prepare.cpp +++ b/Compiler/Core/CorePrep/Prepare.cpp @@ -16,9 +16,14 @@ namespace Visual::XSharp::Core::CorePrep struct State final { - SymbolId nextTemporary{ 1U }; + // One counter names every generated symbol of the module: + // temporaries, condition and short-circuit slots, and lifted + // closure functions. It starts above every identity in the + // whole module and is never reset between functions, because + // identities are module-wide and a per-function start would + // reuse another function's symbols. + SymbolId nextSymbol{ 1U }; Prepared::BlockId nextBlock{ 1U }; - SymbolId nextFunction{ 1U }; std::vector pendingFunctions; // Innermost first; every pair stores (break exit, continue target). std::vector> @@ -157,7 +162,7 @@ namespace Visual::XSharp::Core::CorePrep [[nodiscard]] auto Temporary(std::u32string prefix) -> SymbolName { - const auto id = state.nextTemporary++; + const auto id = state.nextSymbol++; const auto digits = std::to_string(id); prefix.append(digits.begin(), digits.end()); return SymbolName{ id, std::move(prefix) }; @@ -294,7 +299,7 @@ namespace Visual::XSharp::Core::CorePrep { Prepared::Function function; std::vector pendingFunctions; - SymbolId nextFunction{}; + SymbolId nextSymbol{}; }; [[nodiscard]] auto @@ -442,7 +447,7 @@ namespace Visual::XSharp::Core::CorePrep { if (!expression.closureBody) std::abort(); - const auto closureId = cursor.state.nextFunction++; + const auto closureId = cursor.state.nextSymbol++; const auto digits = std::to_string(closureId); std::u32string spelling = U"$closure"; spelling.append(digits.begin(), digits.end()); @@ -776,21 +781,16 @@ namespace Visual::XSharp::Core::CorePrep } [[nodiscard]] auto - PrepareFunction(const Function &function, SymbolId nextFunction) + PrepareFunction(const Function &function, SymbolId nextSymbol) -> PreparedFunctionResult { - SymbolId highest = HighestFunctionSymbol(function); std::vector parameters; parameters.reserve(function.parameters.size()); for (const auto ¶meter : function.parameters) { parameters.push_back({ parameter.symbol, parameter.type }); } - Cursor cursor{ State{ highest + 1U, 1U, nextFunction, {}, {} }, - 0U, - {}, - {}, - true }; + Cursor cursor{ State{ nextSymbol, 1U, {}, {} }, 0U, {}, {}, true }; PrepareStatements(cursor, function.body); // Core verification proves every path returns. A body that still // falls off its end is marked instead of given an invented value. @@ -804,7 +804,7 @@ namespace Visual::XSharp::Core::CorePrep 0U, std::move(cursor.closed) }, std::move(cursor.state.pendingFunctions), - cursor.state.nextFunction, + cursor.state.nextSymbol, }; } } // namespace @@ -819,20 +819,23 @@ namespace Visual::XSharp::Core::CorePrep pending.reserve(module.functions.size()); for (const auto &function : module.functions) pending.emplace_back(function, function.sourceFile); - SymbolId nextFunction = 1U; + // Lifted closure bodies only mention identities already counted + // here or generated by the shared counter, so the seed stays valid + // for functions appended to the queue later. + SymbolId nextSymbol = 1U; for (const auto &[function, _] : pending) - nextFunction - = std::max(nextFunction, HighestFunctionSymbol(function) + 1U); + nextSymbol + = std::max(nextSymbol, HighestFunctionSymbol(function) + 1U); // Closure bodies are lifted as ordinary Core functions and fed back // through the same work queue. This naturally handles nested closures // without adding a second, subtly different lowering implementation. for (std::size_t index = 0U; index < pending.size(); ++index) { - auto result = PrepareFunction(pending[index].first, nextFunction); + auto result = PrepareFunction(pending[index].first, nextSymbol); result.function.sourceFile = pending[index].second; prepared.functions.push_back(std::move(result.function)); - nextFunction = result.nextFunction; + nextSymbol = result.nextSymbol; for (auto &function : result.pendingFunctions) pending.emplace_back(std::move(function), pending[index].second); diff --git a/Compiler/Core/Tests/BUILD.bazel b/Compiler/Core/Tests/BUILD.bazel index 785839bd..50b1cc03 100644 --- a/Compiler/Core/Tests/BUILD.bazel +++ b/Compiler/Core/Tests/BUILD.bazel @@ -18,6 +18,7 @@ cc_binary( "CorePipelineTests.cpp", "LoopLoweringTests.cpp", "ShortCircuitLoweringTests.cpp", + "SymbolAllocationTests.cpp", "TemplateTests.cpp", ], deps = [ diff --git a/Compiler/Core/Tests/SymbolAllocationTests.cpp b/Compiler/Core/Tests/SymbolAllocationTests.cpp new file mode 100644 index 00000000..a7a18f6b --- /dev/null +++ b/Compiler/Core/Tests/SymbolAllocationTests.cpp @@ -0,0 +1,197 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Core/CorePrep/Prepare.hpp" +#include "Visual/XSharp/Core/CorePrep/Verifier.hpp" +#include "Visual/XSharp/Core/Verifier.hpp" +#include "Visual/XSharp/Core/Wire.hpp" +#include "Visual/XSharp/Pipeline.hpp" + +// Symbol identities are unique across a whole module, and CorePrep +// verification checks that one identity never carries two spellings. The +// adapter's generated temporaries and lifted closure names must therefore be +// allocated above every identity of the module from one shared counter, not +// above the identities of the function being lowered. These cases were found +// by the source-to-LLVM fuzz campaign as a verified Core module that the +// native pipeline then rejected. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Prepared = visual_xsharp::core; + + [[nodiscard]] auto + Integer(std::int64_t value) -> Core::Expression + { + return Core::Expression::Constant(value, Core::Type::int64()); + } + + [[nodiscard]] auto + Variable(std::uint64_t id, std::u32string spelling) -> Core::Expression + { + return Core::Expression::Variable({ id, std::move(spelling) }, + Core::Type::int64()); + } + + /// `int result = 8; if (result > 4) { result = result - 2; } return + /// result;` The comparison and the subtraction each need a generated + /// temporary. + [[nodiscard]] auto + Branching(std::uint64_t function, std::uint64_t local, std::u32string name) + -> Core::Function + { + return { + { function, std::move(name) }, + {}, + Core::Type::int64(), + { Core::Statement::Bind({ { local, U"result" }, + Core::Type::int64(), + true, + Integer(8) }), + Core::Statement::If( + Core::Expression::InvokePrimitive( + Core::Primitive::GreaterThan, + { Variable(local, U"result"), Integer(4) }, + Core::Type::boolean()), + { Core::Statement::Assign( + { local, U"result" }, + Core::Expression::InvokePrimitive( + Core::Primitive::Subtract, + { Variable(local, U"result"), Integer(2) }, + Core::Type::int64())) }, + {}), + Core::Statement::Return(Variable(local, U"result")) }, + }; + } + + /// Every identity in the prepared module with each spelling it carries. + void + Record(std::map> &seen, + const Prepared::SymbolName &symbol) + { + if (symbol.id == 0U) + return; + auto &spellings = seen[symbol.id]; + if (std::ranges::find(spellings, symbol.spelling) == spellings.end()) + spellings.push_back(symbol.spelling); + } + + void + RequireUniqueIdentities(const Prepared::CorePrepModule &module) + { + std::map> seen; + for (const auto &function : module.functions) + { + Record(seen, function.symbol); + for (const auto ¶meter : function.parameters) + Record(seen, parameter.symbol); + for (const auto &block : function.blocks) + for (const auto &instruction : block.instructions) + { + Record(seen, instruction.destination); + Record(seen, instruction.closure_function); + for (const auto &operand : instruction.operands) + if (operand.kind == Prepared::Atom::Kind::Variable) + Record(seen, operand.symbol); + } + } + for (const auto &[id, spellings] : seen) + { + CAPTURE(id); + CHECK(spellings.size() == 1U); + } + } + + void + RequireVerifiedPipeline(const Core::Module &module) + { + for (const auto &issue : Core::Verify(module)) + FAIL_CHECK("Core " << issue.code << ": " << issue.message); + REQUIRE(Core::Verify(module).empty()); + const auto prepared = Core::CorePrep::Prepare(module); + for (const auto &issue : Prepared::verify(prepared)) + FAIL_CHECK("CorePrep " << issue.code << ": " << issue.message); + CHECK(Prepared::verify(prepared).empty()); + RequireUniqueIdentities(prepared); + const auto encoded = Core::Wire::Encode(module); + REQUIRE(encoded); + const auto pipeline + = Visual::XSharp::Pipeline::ConsumeCore(encoded.bytes); + CHECK(pipeline); + CHECK(pipeline.llvm); + } +} // namespace + +TEST_CASE("temporaries of one function do not reuse another function's " + "identities", + "[coreprep][symbols]") +{ + // Identities 1 and 2 belong to the first function. Its first temporary + // would be 3 if allocation only looked at that function, which is the + // second function's own name. + RequireVerifiedPipeline( + { { U"Symbols" }, + { Branching(1U, 2U, U"First"), Branching(3U, 4U, U"Second") } }); +} + +TEST_CASE("temporaries stay unique when a later function has lower " + "identities", + "[coreprep][symbols]") +{ + RequireVerifiedPipeline( + { { U"Symbols" }, + { Branching(10U, 11U, U"First"), Branching(1U, 2U, U"Second") } }); +} + +TEST_CASE("lifted closure names and temporaries share one identity counter", + "[coreprep][symbols]") +{ + // The closure is lifted to a generated function symbol while the + // enclosing body also needs temporaries for its arithmetic. + const auto closureType + = Core::Type::function({ Core::Type::int64() }, Core::Type::int64()); + Core::Function function{ + { 1U, U"Evaluate" }, + {}, + Core::Type::int64(), + { Core::Statement::Bind( + { { 2U, U"base" }, Core::Type::int64(), false, Integer(5) }), + Core::Statement::Bind( + { { 3U, U"scale" }, + closureType, + false, + Core::Expression::Closure( + {}, + { { { 4U, U"value" }, Core::Type::int64() } }, + Core::Type::int64(), + { Core::Statement::Return(Core::Expression::InvokePrimitive( + Core::Primitive::Multiply, + { Core::Expression::InvokePrimitive( + Core::Primitive::Add, + { Variable(4U, U"value"), Integer(1) }, + Core::Type::int64()), + Integer(2) }, + Core::Type::int64())) }, + closureType) }), + Core::Statement::Return(Core::Expression::InvokePrimitive( + Core::Primitive::Add, + { Core::Expression::InvokePrimitive( + Core::Primitive::Multiply, + { Variable(2U, U"base"), Integer(3) }, + Core::Type::int64()), + Core::Expression::Apply( + Core::Expression::Variable({ 3U, U"scale" }, closureType), + { Integer(4) }, + Core::Type::int64()) }, + Core::Type::int64())) }, + }; + RequireVerifiedPipeline({ { U"Symbols" }, { std::move(function) } }); +} diff --git a/Compiler/Fuzzing/Corpus/source/cross-function-temporaries.seed b/Compiler/Fuzzing/Corpus/source/cross-function-temporaries.seed new file mode 100644 index 00000000..f727f2c0 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/source/cross-function-temporaries.seed @@ -0,0 +1,2 @@ +namespace Fuzz; class Progrfm { public static long E6aluate() { long result = 8; if (result +> 4) { result = result - 2; } return result; } } class Program { public static long Evannnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnluate() { return 0 + 2 * 3; } } diff --git a/Compiler/Fuzzing/Corpus/source/reserved-llvm-namespace.seed b/Compiler/Fuzzing/Corpus/source/reserved-llvm-namespace.seed new file mode 100644 index 00000000..380b2fc3 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/source/reserved-llvm-namespace.seed @@ -0,0 +1 @@ +namespace llvm.FLLLLLLLQQQQ1QQQQQQQQQQQQQQQQQQQQQQQQQQQQQQQQQQQQQQLLLLuzz; class Program { public static long Evaluate() { long result = 6; if (result >3) { result = result - 2; } return result; } } diff --git a/Documents/CORE-IR.md b/Documents/CORE-IR.md index 5ae93a33..f719d92e 100644 --- a/Documents/CORE-IR.md +++ b/Documents/CORE-IR.md @@ -271,6 +271,15 @@ same Boolean. A CorePrep, Xpp, or Xmm verifier cannot reject a wrong back-edge by itself: a block that jumps to itself is a well-formed control-flow graph. +Both lowerings also allocate generated symbols the same way. Temporaries, +condition and short-circuit slots, and lifted closure names come from one +counter that starts above every symbol identity in the whole module and is +never reset between functions. Identities are module-wide and CorePrep +verification rejects one identity with two spellings, so a counter seeded +from a single function would reuse another function's symbols. +`SymbolAllocationTests.cpp` covers this for multi-function modules and +closures. + ## Expressions Every Core expression has a statically queryable type. diff --git a/Documents/LLVM-BACKEND.md b/Documents/LLVM-BACKEND.md index 240cbe4d..3766c811 100644 --- a/Documents/LLVM-BACKEND.md +++ b/Documents/LLVM-BACKEND.md @@ -61,6 +61,15 @@ verifier require one condition shape. ## Calls and entry bridge +A function's LLVM symbol is its dotted module name, its source spelling, and +its symbol identity, for example `Demo.Main.7`. LLVM reserves every +global name that begins with `llvm.` for intrinsics and rejects a module that +defines one, while `llvm` is an ordinary namespace name in source. Such a +symbol is emitted with a leading `$`, which cannot occur in a source +identifier, for example `$llvm.Main.7`. Declarations in other +per-source objects use the same rule, so cross-object references still +resolve. + Direct calls resolve a stable function symbol to a declared function. Parameter count/types and result type are checked in Xmm before LLVM call construction. A function symbol is not encoded as an integer or ordinary virtual register. From 6bb7e8a2f186383b99f262a63524e154d6af7149 Mon Sep 17 00:00:00 2001 From: Leitwolf11 Date: Fri, 2 Oct 2026 01:18:32 +0300 Subject: [PATCH 5/6] Link Windows JIT objects into one contiguous reservation A hosted AddressSanitizer run aborted the differential smoke with 'IMAGE_REL_AMD64_ADDR32NB relocation requires an ordered section layout' while the same commit passed in a parallel run. On Windows LLJIT links with RuntimeDyld, whose default memory manager maps code and data sections separately. Win64 unwind tables use image-relative relocations that can only be applied when no section of an object lies below the lowest one, so the operating system's choice of addresses decided whether linking succeeded. The ORC session now gives each object a single pre-reserved block on COFF hosts and keeps LLJIT's COFF symbol-flag handling; other hosts keep the default linking layer. The failure depends on address layout and was not reproduced locally, so a test samples 400 independent sessions and the interactive guide states that the evidence for the fix is indirect. --- Compiler/Backend/LLVM/JitSession.cpp | 36 ++++++++++++++++++- .../Backend/LLVM/Tests/JitSessionTests.cpp | 18 ++++++++++ Documents/INTERACTIVE.md | 7 ++++ 3 files changed, 60 insertions(+), 1 deletion(-) diff --git a/Compiler/Backend/LLVM/JitSession.cpp b/Compiler/Backend/LLVM/JitSession.cpp index 44539e91..a27529a7 100644 --- a/Compiler/Backend/LLVM/JitSession.cpp +++ b/Compiler/Backend/LLVM/JitSession.cpp @@ -7,7 +7,9 @@ #include #include #include +#include #include +#include #include #include #include @@ -15,6 +17,8 @@ #include #include #include +#include +#include #include #include #include @@ -262,7 +266,37 @@ namespace Visual::XSharp::Backend::LLVM return; } - auto created = llvm::orc::LLJITBuilder().create(); + llvm::orc::LLJITBuilder builder; + if (llvm::Triple(llvm::sys::getProcessTriple()).isOSBinFormatCOFF()) + { + // Win64 unwind tables use image-relative relocations, which + // RuntimeDyld can only apply when no section of an object + // lies below the lowest one it has already seen. The default + // memory manager maps code and data sections separately, so + // the host decides their order and some address layouts + // abort linking with "relocation requires an ordered section + // layout". Reserving one contiguous block per object fixes + // the order: every section is carved from that reservation. + builder.setObjectLinkingLayerCreator( + [](llvm::orc::ExecutionSession &session) + -> llvm::Expected< + std::unique_ptr> { + auto layer = std::make_unique< + llvm::orc::RTDyldObjectLinkingLayer>( + session, + [](const llvm::MemoryBuffer &) { + return std::make_unique< + llvm::SectionMemoryManager>(nullptr, true); + }); + // Same COFF symbol-flag handling as LLJIT's default + // linking layer. + layer->setOverrideObjectFlagsWithResponsibilityFlags( + true); + layer->setAutoClaimResponsibilityForObjectSymbols(true); + return layer; + }); + } + auto created = builder.create(); if (!created) { initializationError diff --git a/Compiler/Backend/LLVM/Tests/JitSessionTests.cpp b/Compiler/Backend/LLVM/Tests/JitSessionTests.cpp index e895eca5..5bedb0bf 100644 --- a/Compiler/Backend/LLVM/Tests/JitSessionTests.cpp +++ b/Compiler/Backend/LLVM/Tests/JitSessionTests.cpp @@ -467,3 +467,21 @@ TEST_CASE( REQUIRE(after); REQUIRE(std::get(after.value->payload) == 73); } + +TEST_CASE("ORC sessions link unwind metadata regardless of section placement", + "[llvm][orc][layout]") +{ + // Win64 objects carry image-relative unwind relocations that are only + // valid when every section of the object lies above its lowest one. The + // host places separately mapped sections at arbitrary addresses, so a + // linker that maps each section on its own fails for some layouts only. + // Many independent sessions sample those layouts; each must link and run. + constexpr int kSessions = 400; + for (int index = 0; index < kSessions; ++index) + { + const auto result + = InvokeConstant(std::int64_t{ index }, Core::Type::int64()); + REQUIRE(result); + REQUIRE(std::get(result.value->payload) == index); + } +} diff --git a/Documents/INTERACTIVE.md b/Documents/INTERACTIVE.md index c1054c32..8e8793c6 100644 --- a/Documents/INTERACTIVE.md +++ b/Documents/INTERACTIVE.md @@ -155,6 +155,13 @@ JIT execution runs in-process. As with any compiler REPL, evaluating untrusted p with the current user's authority. `vxsi` intentionally does not claim to provide a sandbox, process isolation, memory quota, or security boundary. +On Windows the ORC session links each object into one contiguous memory reservation. Win64 unwind tables use +image-relative relocations that the runtime linker can apply only when no section of an object lies below the lowest one; +with separately mapped sections the operating system chooses that order, and some address layouts abort linking with +"relocation requires an ordered section layout". The failure is layout dependent and was observed once in a hosted +AddressSanitizer run; it has not been reproduced locally, so the reservation removes the cause rather than a +demonstrated repeatable failure. Other hosts keep LLJIT's default linking layer. + ## Build and verify The native targets belong to the repository root, matching the feature's ownership tree: From 9c4fcbe0e059e9ea8acfc45810031568b14eac96 Mon Sep 17 00:00:00 2001 From: Leitwolf11 Date: Fri, 2 Oct 2026 02:32:40 +0300 Subject: [PATCH 6/6] Compile the Windows JIT linking layer only on Windows The contiguous-reservation linking layer introduced for COFF hosts used a SectionMemoryManager constructor that only newer LLVM releases provide. The block was selected at run time but compiled on every host, so builds against distribution LLVM releases on macOS, Ubuntu and Fedora failed with 'no matching constructor for initialization of llvm::SectionMemoryManager'. The linking-layer setup and its two headers are now inside a Windows-only preprocessor block. Other hosts compile the same session setup as before that change and keep LLJIT's default linking layer. --- Compiler/Backend/LLVM/JitSession.cpp | 56 ++++++++++++++-------------- 1 file changed, 28 insertions(+), 28 deletions(-) diff --git a/Compiler/Backend/LLVM/JitSession.cpp b/Compiler/Backend/LLVM/JitSession.cpp index a27529a7..32ce6fe7 100644 --- a/Compiler/Backend/LLVM/JitSession.cpp +++ b/Compiler/Backend/LLVM/JitSession.cpp @@ -7,9 +7,7 @@ #include #include #include -#include #include -#include #include #include #include @@ -17,8 +15,6 @@ #include #include #include -#include -#include #include #include #include @@ -27,6 +23,13 @@ #include #include +#ifdef _WIN32 +// Only the Windows linking layer below needs these. Other hosts build against +// distribution LLVM releases whose memory manager has no reservation mode. +# include +# include +#endif + #include "Visual/XSharp/Backend/LLVM.hpp" #include "Visual/XSharp/Core/Scalar.hpp" @@ -267,35 +270,32 @@ namespace Visual::XSharp::Backend::LLVM } llvm::orc::LLJITBuilder builder; - if (llvm::Triple(llvm::sys::getProcessTriple()).isOSBinFormatCOFF()) - { - // Win64 unwind tables use image-relative relocations, which - // RuntimeDyld can only apply when no section of an object - // lies below the lowest one it has already seen. The default - // memory manager maps code and data sections separately, so - // the host decides their order and some address layouts - // abort linking with "relocation requires an ordered section - // layout". Reserving one contiguous block per object fixes - // the order: every section is carved from that reservation. - builder.setObjectLinkingLayerCreator( - [](llvm::orc::ExecutionSession &session) - -> llvm::Expected< - std::unique_ptr> { - auto layer = std::make_unique< - llvm::orc::RTDyldObjectLinkingLayer>( +#ifdef _WIN32 + // Win64 unwind tables use image-relative relocations, which + // RuntimeDyld can only apply when no section of an object + // lies below the lowest one it has already seen. The default + // memory manager maps code and data sections separately, so + // the host decides their order and some address layouts + // abort linking with "relocation requires an ordered section + // layout". Reserving one contiguous block per object fixes + // the order: every section is carved from that reservation. + builder.setObjectLinkingLayerCreator( + [](llvm::orc::ExecutionSession &session) + -> llvm::Expected> { + auto layer + = std::make_unique( session, [](const llvm::MemoryBuffer &) { return std::make_unique< llvm::SectionMemoryManager>(nullptr, true); }); - // Same COFF symbol-flag handling as LLJIT's default - // linking layer. - layer->setOverrideObjectFlagsWithResponsibilityFlags( - true); - layer->setAutoClaimResponsibilityForObjectSymbols(true); - return layer; - }); - } + // Same COFF symbol-flag handling as LLJIT's default + // linking layer. + layer->setOverrideObjectFlagsWithResponsibilityFlags(true); + layer->setAutoClaimResponsibilityForObjectSymbols(true); + return layer; + }); +#endif auto created = builder.create(); if (!created) {