diff --git a/.github/workflows/fuzzing.yml b/.github/workflows/fuzzing.yml index 501f742b..779a65f5 100644 --- a/.github/workflows/fuzzing.yml +++ b/.github/workflows/fuzzing.yml @@ -152,6 +152,11 @@ jobs: env: VXS_FUZZ_CORPUS: ${{ runner.temp }}/vxs-fuzz-corpus run: go run ./helpers/cmd/develop fuzz + - name: Run bounded ThreadSanitizer ownership campaign + shell: bash + env: + VXS_FUZZ_CORPUS: ${{ runner.temp }}/vxs-fuzz-corpus + run: go run ./helpers/cmd/develop fuzz-thread - name: Run nightly ASan and UBSan stress campaign if: github.event_name == 'schedule' shell: bash @@ -216,6 +221,11 @@ jobs: env: VXS_FUZZ_CORPUS: ${{ runner.temp }}/vxs-fuzz-corpus run: go run ./helpers/cmd/develop fuzz + - name: Run bounded ThreadSanitizer ownership campaign + shell: bash + env: + VXS_FUZZ_CORPUS: ${{ runner.temp }}/vxs-fuzz-corpus + run: go run ./helpers/cmd/develop fuzz-thread - name: Run nightly ASan and UBSan stress campaign if: github.event_name == 'schedule' shell: bash diff --git a/.github/workflows/language-layers.yml b/.github/workflows/language-layers.yml index 9fb2874c..a3781254 100644 --- a/.github/workflows/language-layers.yml +++ b/.github/workflows/language-layers.yml @@ -54,6 +54,11 @@ jobs: working-directory: Compiler run: cabal test visual-xsharp-compiler-tests + - name: Test frontend fuzz feedback and mutation policy + shell: pwsh + working-directory: Compiler + run: cabal test fuzz-feedback-tests + - name: Check Haskell package metadata shell: pwsh run: | diff --git a/.gitignore b/.gitignore index 498ca605..2c1efa36 100644 --- a/.gitignore +++ b/.gitignore @@ -6,6 +6,7 @@ bin/ xide_api/ xide/ .codex/ +.claude/ .agents/ .gradle/ .kotlin/ @@ -13,6 +14,8 @@ xide/ node_modules/ dist/ dist-newstyle/ +dist-fuzz-coverage/ +*.tix coverage/ *.profraw *.profdata diff --git a/Compiler/Backend/LLVM/Codegen.cpp b/Compiler/Backend/LLVM/Codegen.cpp index e06528b0..a53299d4 100644 --- a/Compiler/Backend/LLVM/Codegen.cpp +++ b/Compiler/Backend/LLVM/Codegen.cpp @@ -131,6 +131,13 @@ namespace Visual::XSharp::Backend::LLVM result->append(*spelling); result->append("."); result->append(std::to_string(function.symbol.id)); + // LLVM reserves every global name beginning with `llvm.` for + // intrinsics and rejects a module that defines one, but `llvm` + // is an ordinary namespace name in source. `$` cannot occur in + // a source identifier, so the prefixed name is unambiguous and + // cannot collide with another module's symbol. + if (result->starts_with("llvm.")) + result->insert(0U, 1U, '$'); return result; } diff --git a/Compiler/Backend/LLVM/JitSession.cpp b/Compiler/Backend/LLVM/JitSession.cpp index 44539e91..32ce6fe7 100644 --- a/Compiler/Backend/LLVM/JitSession.cpp +++ b/Compiler/Backend/LLVM/JitSession.cpp @@ -23,6 +23,13 @@ #include #include +#ifdef _WIN32 +// Only the Windows linking layer below needs these. Other hosts build against +// distribution LLVM releases whose memory manager has no reservation mode. +# include +# include +#endif + #include "Visual/XSharp/Backend/LLVM.hpp" #include "Visual/XSharp/Core/Scalar.hpp" @@ -262,7 +269,34 @@ namespace Visual::XSharp::Backend::LLVM return; } - auto created = llvm::orc::LLJITBuilder().create(); + llvm::orc::LLJITBuilder builder; +#ifdef _WIN32 + // Win64 unwind tables use image-relative relocations, which + // RuntimeDyld can only apply when no section of an object + // lies below the lowest one it has already seen. The default + // memory manager maps code and data sections separately, so + // the host decides their order and some address layouts + // abort linking with "relocation requires an ordered section + // layout". Reserving one contiguous block per object fixes + // the order: every section is carved from that reservation. + builder.setObjectLinkingLayerCreator( + [](llvm::orc::ExecutionSession &session) + -> llvm::Expected> { + auto layer + = std::make_unique( + session, + [](const llvm::MemoryBuffer &) { + return std::make_unique< + llvm::SectionMemoryManager>(nullptr, true); + }); + // Same COFF symbol-flag handling as LLJIT's default + // linking layer. + layer->setOverrideObjectFlagsWithResponsibilityFlags(true); + layer->setAutoClaimResponsibilityForObjectSymbols(true); + return layer; + }); +#endif + auto created = builder.create(); if (!created) { initializationError diff --git a/Compiler/Backend/LLVM/Tests/BUILD.bazel b/Compiler/Backend/LLVM/Tests/BUILD.bazel index 7bd32f56..4853bfcc 100644 --- a/Compiler/Backend/LLVM/Tests/BUILD.bazel +++ b/Compiler/Backend/LLVM/Tests/BUILD.bazel @@ -8,6 +8,9 @@ cc_binary( "CallableInvocationTests.cpp", "LLVMBackendTests.cpp", "JitSessionTests.cpp", + "LoopExecutionTests.cpp", + "ReservedSymbolTests.cpp", + "ShortCircuitExecutionTests.cpp", ], deps = [ "//Compiler/Backend/LLVM:llvm_backend", diff --git a/Compiler/Backend/LLVM/Tests/JitSessionTests.cpp b/Compiler/Backend/LLVM/Tests/JitSessionTests.cpp index e895eca5..5bedb0bf 100644 --- a/Compiler/Backend/LLVM/Tests/JitSessionTests.cpp +++ b/Compiler/Backend/LLVM/Tests/JitSessionTests.cpp @@ -467,3 +467,21 @@ TEST_CASE( REQUIRE(after); REQUIRE(std::get(after.value->payload) == 73); } + +TEST_CASE("ORC sessions link unwind metadata regardless of section placement", + "[llvm][orc][layout]") +{ + // Win64 objects carry image-relative unwind relocations that are only + // valid when every section of the object lies above its lowest one. The + // host places separately mapped sections at arbitrary addresses, so a + // linker that maps each section on its own fails for some layouts only. + // Many independent sessions sample those layouts; each must link and run. + constexpr int kSessions = 400; + for (int index = 0; index < kSessions; ++index) + { + const auto result + = InvokeConstant(std::int64_t{ index }, Core::Type::int64()); + REQUIRE(result); + REQUIRE(std::get(result.value->payload) == index); + } +} diff --git a/Compiler/Backend/LLVM/Tests/LoopExecutionTests.cpp b/Compiler/Backend/LLVM/Tests/LoopExecutionTests.cpp new file mode 100644 index 00000000..e38cabaa --- /dev/null +++ b/Compiler/Backend/LLVM/Tests/LoopExecutionTests.cpp @@ -0,0 +1,400 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Backend/LLVM.hpp" +#include "Visual/XSharp/Core/IR.hpp" +#include "Visual/XSharp/Core/Wire.hpp" +#include "Visual/XSharp/Pipeline.hpp" + +// Executable loop regressions. Every case is lowered from structured Core +// through CorePrep, Xpp, Xmm and LLVM, run in ORC, and compared with a value +// computed by an ordinary host loop written in this file. Comparing the +// optimized and unoptimized pipelines with each other is not sufficient: a +// defect in the shared Core-to-CorePrep adapter makes both agree on the same +// wrong answer, or on the same non-terminating program. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Llvm = Visual::XSharp::Backend::LLVM; + namespace Pipeline = Visual::XSharp::Pipeline; + namespace Prepared = visual_xsharp::core; + + constexpr std::uint64_t kTotal = 2U; + constexpr std::uint64_t kIndex = 3U; + constexpr std::uint64_t kInner = 4U; + constexpr std::int64_t kLimits = 13; + + [[nodiscard]] auto + Spelling(std::uint64_t id) -> std::u32string + { + return id == kTotal ? U"total" : id == kIndex ? U"index" : U"inner"; + } + + [[nodiscard]] auto + Integer(std::int64_t value) -> Core::Expression + { + return Core::Expression::Constant(value, Core::Type::int64()); + } + + [[nodiscard]] auto + Variable(std::uint64_t id) -> Core::Expression + { + return Core::Expression::Variable({ id, Spelling(id) }, + Core::Type::int64()); + } + + [[nodiscard]] auto + Compare(Core::Primitive operation, std::uint64_t id, std::int64_t value) + -> Core::Expression + { + return Core::Expression::InvokePrimitive( + operation, + { Variable(id), Integer(value) }, + Core::Type::boolean()); + } + + [[nodiscard]] auto + Add(std::uint64_t destination, Core::Expression value) -> Core::Statement + { + return Core::Statement::Assign( + { destination, Spelling(destination) }, + Core::Expression::InvokePrimitive( + Core::Primitive::Add, + { Variable(destination), std::move(value) }, + Core::Type::int64())); + } + + [[nodiscard]] auto + Increment(std::uint64_t id) -> Core::Statement + { + return Add(id, Integer(1)); + } + + [[nodiscard]] auto + When(std::uint64_t id, std::int64_t value, Core::Statement then) + -> Core::Statement + { + return Core::Statement::If(Compare(Core::Primitive::Equal, id, value), + { std::move(then) }, + {}); + } + + [[nodiscard]] auto + Declare(std::uint64_t id) -> Core::Statement + { + return Core::Statement::Bind( + { { id, Spelling(id) }, Core::Type::int64(), true, Integer(0) }); + } + + /// `int total = 0; int index = 0; ; return total;` + [[nodiscard]] auto + LoopModule(Core::Statement loop) -> Core::Module + { + Core::Function function{ + { 1U, U"Evaluate" }, + {}, + Core::Type::int64(), + { Declare(kTotal), + Declare(kIndex), + std::move(loop), + Core::Statement::Return(Variable(kTotal)) }, + }; + return { { U"Loops" }, { std::move(function) } }; + } + + /// True when every CorePrep block can still reach a function return. + /// A loop region that cannot leave is rejected before it is executed, so + /// a control-flow regression fails this suite instead of hanging it. + [[nodiscard]] auto + EveryBlockReachesReturn(const Prepared::Function &function) -> bool + { + const auto successors = [](const Prepared::Block &block) { + std::vector targets; + if (block.terminator.kind == Prepared::Terminator::Kind::Jump) + targets = { block.terminator.true_target }; + if (block.terminator.kind == Prepared::Terminator::Kind::Branch) + targets = { block.terminator.true_target, + block.terminator.false_target }; + return targets; + }; + // Backward fixed point from the return blocks. + std::vector returning; + for (const auto &block : function.blocks) + if (block.terminator.kind == Prepared::Terminator::Kind::Return) + returning.push_back(block.id); + for (bool changed = true; changed;) + { + changed = false; + for (const auto &block : function.blocks) + { + if (std::ranges::find(returning, block.id) != returning.end()) + continue; + const auto targets = successors(block); + if (std::ranges::any_of(targets, [&](const auto target) { + return std::ranges::find(returning, target) + != returning.end(); + })) + { + returning.push_back(block.id); + changed = true; + } + } + } + return returning.size() == function.blocks.size(); + } + + [[nodiscard]] auto + Run(const Core::Module &module, bool optimize) + -> std::optional + { + const auto encoded = Core::Wire::Encode(module); + REQUIRE(encoded); + Pipeline::Options options; + options.optimize_xpp = optimize; + options.optimize_xmm = optimize; + options.llvm.optimization = optimize ? Llvm::OptimizationLevel::Default + : Llvm::OptimizationLevel::Debug; + const auto pipeline = Pipeline::ConsumeCore(encoded.bytes, options); + REQUIRE(pipeline); + REQUIRE(pipeline.llvm); + REQUIRE(pipeline.core_prep); + REQUIRE(pipeline.core_prep->functions.size() == 1U); + REQUIRE(EveryBlockReachesReturn(pipeline.core_prep->functions.front())); + + constexpr std::string_view kSymbol = "Loops.Evaluate.1"; + Llvm::JitSession session; + const auto rejected = session.AddModule(pipeline.llvm->bitcode, + "loop-execution", + kSymbol, + Core::Type::int64()); + REQUIRE_FALSE(rejected); + const auto result = session.InvokeScalar(kSymbol, Core::Type::int64()); + REQUIRE(result); + return std::get(result.value->payload); + } + + void + CheckBothPipelines(const Core::Module &module, std::int64_t expected) + { + CHECK(Run(module, false) == expected); + CHECK(Run(module, true) == expected); + } +} // namespace + +TEST_CASE("for-loop runs its update once per iteration and then re-tests", + "[llvm][loop][execution]") +{ + for (std::int64_t limit = 0; limit < kLimits; ++limit) + { + std::int64_t expected{}; + for (std::int64_t index = 0; index < limit; ++index) + expected += index; + CAPTURE(limit); + CheckBothPipelines( + LoopModule(Core::Statement::For( + Compare(Core::Primitive::LessThan, kIndex, limit), + { Add(kTotal, Variable(kIndex)) }, + { Increment(kIndex) })), + expected); + } +} + +TEST_CASE("for-loop continue still updates and break skips the update", + "[llvm][loop][execution]") +{ + for (std::int64_t limit = 0; limit < kLimits; ++limit) + { + std::int64_t expected{}; + for (std::int64_t index = 0; index < limit; ++index) + { + if (index == 2) + continue; + if (index == 9) + break; + expected += index; + } + CAPTURE(limit); + CheckBothPipelines( + LoopModule(Core::Statement::For( + Compare(Core::Primitive::LessThan, kIndex, limit), + { When(kIndex, 2, Core::Statement::Continue()), + When(kIndex, 9, Core::Statement::Break()), + Add(kTotal, Variable(kIndex)) }, + { Increment(kIndex) })), + expected); + } +} + +TEST_CASE("for-loop observes the index value left by break and by exhaustion", + "[llvm][loop][execution]") +{ + // Returning the induction variable distinguishes "update ran after the + // last body" from "update skipped", which a sum of indices cannot. + for (std::int64_t limit = 0; limit < kLimits; ++limit) + { + std::int64_t index = 0; + for (; index < limit; ++index) + if (index == 5) + break; + auto module = LoopModule(Core::Statement::For( + Compare(Core::Primitive::LessThan, kIndex, limit), + { When(kIndex, 5, Core::Statement::Break()) }, + { Increment(kIndex) })); + module.functions.front().body.back() + = Core::Statement::Return(Variable(kIndex)); + CAPTURE(limit); + CheckBothPipelines(module, index); + } +} + +TEST_CASE("for-loop accepts a numeric condition and an empty update", + "[llvm][loop][execution]") +{ + std::int64_t expected{}; + for (std::int64_t index = 0; 3 - index; ++index) + expected += index; + CheckBothPipelines( + LoopModule(Core::Statement::For( + Core::Expression::InvokePrimitive(Core::Primitive::Subtract, + { Integer(3), Variable(kIndex) }, + Core::Type::int64()), + { Add(kTotal, Variable(kIndex)) }, + { Increment(kIndex) })), + expected); + + expected = 0; + for (std::int64_t index = 0; index < 4;) + { + expected += index; + ++index; + } + CheckBothPipelines(LoopModule(Core::Statement::For( + Compare(Core::Primitive::LessThan, kIndex, 4), + { Add(kTotal, Variable(kIndex)), Increment(kIndex) }, + {})), + expected); +} + +TEST_CASE("while-loop does not re-run statements that precede it", + "[llvm][loop][execution]") +{ + for (std::int64_t limit = 0; limit < kLimits; ++limit) + { + std::int64_t expected{}; + std::int64_t index{}; + while (index < limit) + { + if (index == 1) + { + ++index; + continue; + } + if (index == 7) + break; + expected += index; + ++index; + } + CAPTURE(limit); + CheckBothPipelines( + LoopModule(Core::Statement::While( + Compare(Core::Primitive::LessThan, kIndex, limit), + { Core::Statement::If( + Compare(Core::Primitive::Equal, kIndex, 1), + { Increment(kIndex), Core::Statement::Continue() }, + {}), + When(kIndex, 7, Core::Statement::Break()), + Add(kTotal, Variable(kIndex)), + Increment(kIndex) })), + expected); + } +} + +TEST_CASE("do-while runs its body before the first test and on continue", + "[llvm][loop][execution]") +{ + for (std::int64_t limit = 0; limit < kLimits; ++limit) + { + std::int64_t expected{}; + std::int64_t index{}; + do + { + ++index; + if (index == 2) + continue; + if (index == 5) + break; + expected += index; + } while (index < limit); + CAPTURE(limit); + CheckBothPipelines( + LoopModule(Core::Statement::DoWhile( + { Increment(kIndex), + When(kIndex, 2, Core::Statement::Continue()), + When(kIndex, 5, Core::Statement::Break()), + Add(kTotal, Variable(kIndex)) }, + Compare(Core::Primitive::LessThan, kIndex, limit))), + expected); + } +} + +TEST_CASE("nested loops transfer only within their own loop", + "[llvm][loop][execution]") +{ + for (std::int64_t limit = 0; limit < 6; ++limit) + { + std::int64_t expected{}; + for (std::int64_t index = 0; index < limit; ++index) + { + if (index == 3) + continue; + for (std::int64_t inner = 0; inner < 4; ++inner) + { + if (inner == 1) + continue; + if (inner == 3) + break; + expected += index + inner; + } + std::int64_t tail{}; + while (tail < 2) + { + expected += 100; + ++tail; + } + } + // The inner counters reuse `inner`; rebinding per outer iteration + // is expressed as an assignment so each binding is defined once. + auto reset + = Core::Statement::Assign({ kInner, Spelling(kInner) }, Integer(0)); + auto module = LoopModule(Core::Statement::For( + Compare(Core::Primitive::LessThan, kIndex, limit), + { When(kIndex, 3, Core::Statement::Continue()), + reset, + Core::Statement::For( + Compare(Core::Primitive::LessThan, kInner, 4), + { When(kInner, 1, Core::Statement::Continue()), + When(kInner, 3, Core::Statement::Break()), + Add(kTotal, Variable(kIndex)), + Add(kTotal, Variable(kInner)) }, + { Increment(kInner) }), + reset, + Core::Statement::While( + Compare(Core::Primitive::LessThan, kInner, 2), + { Add(kTotal, Integer(100)), Increment(kInner) }) }, + { Increment(kIndex) })); + auto &body = module.functions.front().body; + body.insert(body.begin() + 2, Declare(kInner)); + CAPTURE(limit); + CheckBothPipelines(module, expected); + } +} diff --git a/Compiler/Backend/LLVM/Tests/ReservedSymbolTests.cpp b/Compiler/Backend/LLVM/Tests/ReservedSymbolTests.cpp new file mode 100644 index 00000000..03bc6583 --- /dev/null +++ b/Compiler/Backend/LLVM/Tests/ReservedSymbolTests.cpp @@ -0,0 +1,87 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Backend/LLVM.hpp" +#include "Visual/XSharp/Core/IR.hpp" +#include "Visual/XSharp/Core/Wire.hpp" +#include "Visual/XSharp/Pipeline.hpp" + +// LLVM reserves every global name that begins with `llvm.` for intrinsics +// and rejects a module that defines one. `llvm` is an ordinary Visual X# +// namespace name, so the backend must give such functions a symbol outside +// the reserved prefix instead of failing module verification. Found by the +// source-to-LLVM fuzz campaign. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Llvm = Visual::XSharp::Backend::LLVM; + namespace Pipeline = Visual::XSharp::Pipeline; + + [[nodiscard]] auto + Constant(std::vector moduleName, std::int64_t value) + -> Core::Module + { + return { + std::move(moduleName), + { Core::Function{ + { 1U, U"Evaluate" }, + {}, + Core::Type::int64(), + { Core::Statement::Return( + Core::Expression::Constant(value, Core::Type::int64())) }, + } } + }; + } + + [[nodiscard]] auto + Invoke(const Core::Module &module, std::string_view symbol) -> std::int64_t + { + const auto encoded = Core::Wire::Encode(module); + REQUIRE(encoded); + const auto pipeline = Pipeline::ConsumeCore(encoded.bytes); + if (pipeline.llvm_error) + FAIL_CHECK(pipeline.llvm_error->code + << ": " << pipeline.llvm_error->message); + REQUIRE(pipeline); + REQUIRE(pipeline.llvm); + // LLVM quotes a global name that contains `$`. + const auto quoted = "@\"" + std::string(symbol) + "\"("; + const auto plain = "@" + std::string(symbol) + "("; + CHECK((pipeline.llvm->llvm_ir.find(quoted) != std::string::npos + || pipeline.llvm->llvm_ir.find(plain) != std::string::npos)); + + Llvm::JitSession session; + REQUIRE_FALSE(session.AddModule(pipeline.llvm->bitcode, + "reserved-symbol", + symbol, + Core::Type::int64())); + const auto result = session.InvokeScalar(symbol, Core::Type::int64()); + REQUIRE(result); + return std::get(result.value->payload); + } +} // namespace + +TEST_CASE("functions in a namespace named llvm leave the intrinsic prefix", + "[llvm][symbols]") +{ + CHECK(Invoke(Constant({ U"llvm" }, 11), "$llvm.Evaluate.1") == 11); + CHECK(Invoke(Constant({ U"llvm", U"Fuzz" }, 12), "$llvm.Fuzz.Evaluate.1") + == 12); +} + +TEST_CASE("namespaces that merely resemble the intrinsic prefix are unchanged", + "[llvm][symbols]") +{ + CHECK(Invoke(Constant({ U"llvmx" }, 21), "llvmx.Evaluate.1") == 21); + CHECK(Invoke(Constant({ U"Llvm" }, 22), "Llvm.Evaluate.1") == 22); + CHECK(Invoke(Constant({ U"Demo", U"llvm" }, 23), "Demo.llvm.Evaluate.1") + == 23); +} diff --git a/Compiler/Backend/LLVM/Tests/ShortCircuitExecutionTests.cpp b/Compiler/Backend/LLVM/Tests/ShortCircuitExecutionTests.cpp new file mode 100644 index 00000000..f4f2d5d1 --- /dev/null +++ b/Compiler/Backend/LLVM/Tests/ShortCircuitExecutionTests.cpp @@ -0,0 +1,318 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Backend/LLVM.hpp" +#include "Visual/XSharp/Core/IR.hpp" +#include "Visual/XSharp/Core/Wire.hpp" +#include "Visual/XSharp/Pipeline.hpp" + +// Executable short-circuit regressions. The right operand of each case is +// only well defined when the left operand guards it: a division whose +// divisor the guard excludes, or a recursive call the guard terminates. A +// pipeline that evaluates both operands traps or never returns, so these +// programs observe laziness itself rather than only the final Boolean. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Llvm = Visual::XSharp::Backend::LLVM; + namespace Pipeline = Visual::XSharp::Pipeline; + + constexpr std::uint64_t kValue = 2U; + constexpr std::uint64_t kTotal = 3U; + constexpr std::uint64_t kDown = 10U; + constexpr std::uint64_t kDownParameter = 11U; + + [[nodiscard]] auto + Integer(std::int64_t value) -> Core::Expression + { + return Core::Expression::Constant(value, Core::Type::int64()); + } + + [[nodiscard]] auto + Variable(std::uint64_t id, std::u32string spelling) -> Core::Expression + { + return Core::Expression::Variable({ id, std::move(spelling) }, + Core::Type::int64()); + } + + [[nodiscard]] auto + Value() -> Core::Expression + { + return Variable(kValue, U"value"); + } + + [[nodiscard]] auto + Binary(Core::Primitive operation, + Core::Expression left, + Core::Expression right, + Core::Type type) -> Core::Expression + { + return Core::Expression::InvokePrimitive( + operation, + { std::move(left), std::move(right) }, + std::move(type)); + } + + [[nodiscard]] auto + Compare(Core::Primitive operation, + Core::Expression left, + Core::Expression right) -> Core::Expression + { + return Binary(operation, + std::move(left), + std::move(right), + Core::Type::boolean()); + } + + [[nodiscard]] auto + Quotient(std::int64_t dividend, Core::Expression divisor) + -> Core::Expression + { + return Binary(Core::Primitive::Divide, + Integer(dividend), + std::move(divisor), + Core::Type::int64()); + } + + /// `int value = ; int total = 0; ; return total;` + [[nodiscard]] auto + Evaluate(std::int64_t input, std::vector statements) + -> Core::Function + { + std::vector body{ + Core::Statement::Bind({ { kValue, U"value" }, + Core::Type::int64(), + true, + Integer(input) }), + Core::Statement::Bind( + { { kTotal, U"total" }, Core::Type::int64(), true, Integer(0) }) + }; + body.insert(body.end(), + std::make_move_iterator(statements.begin()), + std::make_move_iterator(statements.end())); + body.push_back(Core::Statement::Return(Variable(kTotal, U"total"))); + return { { 1U, U"Evaluate" }, + {}, + Core::Type::int64(), + std::move(body) }; + } + + [[nodiscard]] auto + SetTotal(std::int64_t value) -> Core::Statement + { + return Core::Statement::Assign({ kTotal, U"total" }, Integer(value)); + } + + [[nodiscard]] auto + Run(const Core::Module &module, bool optimize) -> std::int64_t + { + const auto encoded = Core::Wire::Encode(module); + REQUIRE(encoded); + Pipeline::Options options; + options.optimize_xpp = optimize; + options.optimize_xmm = optimize; + options.llvm.optimization = optimize ? Llvm::OptimizationLevel::Default + : Llvm::OptimizationLevel::Debug; + const auto pipeline = Pipeline::ConsumeCore(encoded.bytes, options); + REQUIRE(pipeline); + REQUIRE(pipeline.llvm); + + constexpr std::string_view kSymbol = "ShortCircuit.Evaluate.1"; + Llvm::JitSession session; + const auto rejected = session.AddModule(pipeline.llvm->bitcode, + "short-circuit-execution", + kSymbol, + Core::Type::int64()); + REQUIRE_FALSE(rejected); + const auto result = session.InvokeScalar(kSymbol, Core::Type::int64()); + REQUIRE(result); + return std::get(result.value->payload); + } + + void + CheckBothPipelines(std::vector functions, + std::int64_t expected) + { + const Core::Module module{ { U"ShortCircuit" }, std::move(functions) }; + CHECK(Run(module, false) == expected); + CHECK(Run(module, true) == expected); + } +} // namespace + +TEST_CASE("logical and does not evaluate a division its left operand excludes", + "[llvm][shortcircuit][execution]") +{ + for (std::int64_t input = -3; input <= 3; ++input) + { + const std::int64_t expected = (input != 0 && 12 / input > 2) ? 1 : 0; + CAPTURE(input); + CheckBothPipelines( + { Evaluate( + input, + { Core::Statement::If( + Compare( + Core::Primitive::LogicalAnd, + Compare(Core::Primitive::NotEqual, Value(), Integer(0)), + Compare(Core::Primitive::GreaterThan, + Quotient(12, Value()), + Integer(2))), + { SetTotal(1) }, + {}) }) }, + expected); + } +} + +TEST_CASE("logical or does not evaluate a division its left operand excludes", + "[llvm][shortcircuit][execution]") +{ + for (std::int64_t input = -3; input <= 3; ++input) + { + const std::int64_t expected = (input == 0 || 12 / input < 0) ? 1 : 0; + CAPTURE(input); + CheckBothPipelines( + { Evaluate( + input, + { Core::Statement::If( + Compare( + Core::Primitive::LogicalOr, + Compare(Core::Primitive::Equal, Value(), Integer(0)), + Compare(Core::Primitive::LessThan, + Quotient(12, Value()), + Integer(0))), + { SetTotal(1) }, + {}) }) }, + expected); + } +} + +TEST_CASE("short-circuit value is usable as an ordinary Boolean binding", + "[llvm][shortcircuit][execution]") +{ + // The operator's value, not only its branch, must be the lazy result: + // bind it, then test the binding after an unrelated statement. + constexpr std::uint64_t kFlag = 4U; + for (std::int64_t input = -2; input <= 2; ++input) + { + const bool flag = input != 0 && 8 / input == 4; + const std::int64_t expected = flag ? 7 : 5; + CAPTURE(input); + CheckBothPipelines( + { Evaluate(input, + { Core::Statement::Bind( + { { kFlag, U"flag" }, + Core::Type::boolean(), + false, + Compare(Core::Primitive::LogicalAnd, + Compare(Core::Primitive::NotEqual, + Value(), + Integer(0)), + Compare(Core::Primitive::Equal, + Quotient(8, Value()), + Integer(4))) }), + SetTotal(5), + Core::Statement::If( + Core::Expression::Variable({ kFlag, U"flag" }, + Core::Type::boolean()), + { SetTotal(7) }, + {}) }) }, + expected); + } +} + +TEST_CASE("guarded recursion terminates through a short-circuit operator", + "[llvm][shortcircuit][execution]") +{ + // bool Down(int n) { return n == 0 || Down(n - 1); } + // Evaluating the call eagerly recurses below zero without bound. + const auto downType + = Core::Type::function({ Core::Type::int64() }, Core::Type::boolean()); + const auto parameter = [] { + return Variable(kDownParameter, U"n"); + }; + Core::Function down{ + { kDown, U"Down" }, + { { { kDownParameter, U"n" }, Core::Type::int64() } }, + Core::Type::boolean(), + { Core::Statement::Return(Compare( + Core::Primitive::LogicalOr, + Compare(Core::Primitive::Equal, parameter(), Integer(0)), + Core::Expression::Apply( + Core::Expression::Variable({ kDown, U"Down" }, downType), + { Binary(Core::Primitive::Subtract, + parameter(), + Integer(1), + Core::Type::int64()) }, + Core::Type::boolean()))) }, + }; + for (std::int64_t input = 0; input <= 6; ++input) + { + CAPTURE(input); + CheckBothPipelines( + { down, + Evaluate(input, + { Core::Statement::If( + Core::Expression::Apply( + Core::Expression::Variable({ kDown, U"Down" }, + downType), + { Value() }, + Core::Type::boolean()), + { SetTotal(1) }, + {}) }) }, + 1); + } +} + +TEST_CASE("short-circuit loop condition guards its own right operand", + "[llvm][shortcircuit][execution]") +{ + // while (value < limit && 100 / (limit - value) > 0) { ... } + // The division is undefined exactly when the left operand is false. + for (std::int64_t limit = 0; limit <= 6; ++limit) + { + std::int64_t expected{}; + std::int64_t value{}; + while (value < limit && 100 / (limit - value) > 0) + { + expected += value; + ++value; + } + const auto remaining = [limit] { + return Binary(Core::Primitive::Subtract, + Integer(limit), + Value(), + Core::Type::int64()); + }; + CAPTURE(limit); + CheckBothPipelines( + { Evaluate( + 0, + { Core::Statement::While( + Compare(Core::Primitive::LogicalAnd, + Compare(Core::Primitive::LessThan, + Value(), + Integer(limit)), + Compare(Core::Primitive::GreaterThan, + Quotient(100, remaining()), + Integer(0))), + { Core::Statement::Assign({ kTotal, U"total" }, + Binary(Core::Primitive::Add, + Variable(kTotal, U"total"), + Value(), + Core::Type::int64())), + Core::Statement::Assign( + { kValue, U"value" }, + Binary(Core::Primitive::Add, + Value(), + Integer(1), + Core::Type::int64())) }) }) }, + expected); + } +} diff --git a/Compiler/Cli/Arguments/Options.hpp b/Compiler/Cli/Arguments/Options.hpp index dfa91ffb..b7e63d55 100644 --- a/Compiler/Cli/Arguments/Options.hpp +++ b/Compiler/Cli/Arguments/Options.hpp @@ -116,6 +116,9 @@ struct CompilerSettings LlvmOptLevel llvmOptLevel; LlvmCompiler llvmCompiler; LlvmLto llvmLto; + + bool + operator==(const CompilerSettings &) const = default; }; struct CliOptions @@ -155,6 +158,9 @@ struct CliOptions bool llvmOptOverride; bool llvmCompilerOverride; bool llvmLtoOverride; + + bool + operator==(const CliOptions &) const = default; }; // Fully resolved values for one compiler invocation. A project evaluation can @@ -183,6 +189,9 @@ struct CliParseOutcome CliOptions options; std::optional helpCommand; std::string diagnostic; + + bool + operator==(const CliParseOutcome &) const = default; }; [[nodiscard]] CompilerSettings diff --git a/Compiler/Cli/Fuzzing/BUILD.bazel b/Compiler/Cli/Fuzzing/BUILD.bazel new file mode 100644 index 00000000..a9176242 --- /dev/null +++ b/Compiler/Cli/Fuzzing/BUILD.bazel @@ -0,0 +1,7 @@ +load("@rules_cc//cc:cc_binary.bzl", "cc_binary") + +cc_binary( + name = "cli_fuzzer", + srcs = ["CliFuzzer.cpp"], + deps = ["//Compiler/Cli/Arguments:arguments", "@llvm//:llvm"], +) diff --git a/Compiler/Cli/Fuzzing/CliFuzzer.cpp b/Compiler/Cli/Fuzzing/CliFuzzer.cpp new file mode 100644 index 00000000..b8071a8b --- /dev/null +++ b/Compiler/Cli/Fuzzing/CliFuzzer.cpp @@ -0,0 +1,54 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include + +#include "Compiler/Cli/Arguments/Options.hpp" + +extern "C" int +LLVMFuzzerTestOneInput(const std::uint8_t *data, std::size_t size) +{ + if (size > 16384U) + return 0; + std::vector arguments{ "vxs" }; + std::string word; + for (const auto byte : std::span(data, size)) + { + if (byte == 0U) + { + arguments.push_back(word); + word.clear(); + if (arguments.size() == 128U) + break; + } + else + word.push_back(static_cast(byte)); + } + if (!word.empty()) + arguments.push_back(word); + const auto original = arguments; + std::vector argv; + for (auto &argument : arguments) + argv.push_back(argument.data()); + const auto first + = ParseCommandLine(static_cast(argv.size()), argv.data()); + const auto second + = ParseCommandLine(static_cast(argv.size()), argv.data()); + // Full typed-model equality checks defaults and override bits as well as + // diagnostics. Parsing must never mutate borrowed argv storage. + if (first != second || arguments != original) + llvm::report_fatal_error( + "CLI parsing is nondeterministic or mutated argv"); + if (first.result == CliParseResult::kError && first.diagnostic.empty()) + llvm::report_fatal_error("CLI rejected input without a diagnostic"); + if (first.result == CliParseResult::kReady + && first.options.command == CliCommand::kNone) + llvm::report_fatal_error( + "CLI accepted input without selecting a command"); + return 0; +} diff --git a/Compiler/Core/CorePrep/Prepare.cpp b/Compiler/Core/CorePrep/Prepare.cpp index f174b4c9..af0078a8 100644 --- a/Compiler/Core/CorePrep/Prepare.cpp +++ b/Compiler/Core/CorePrep/Prepare.cpp @@ -16,38 +16,20 @@ namespace Visual::XSharp::Core::CorePrep struct State final { - SymbolId nextTemporary{ 1U }; + // One counter names every generated symbol of the module: + // temporaries, condition and short-circuit slots, and lifted + // closure functions. It starts above every identity in the + // whole module and is never reset between functions, because + // identities are module-wide and a per-function start would + // reuse another function's symbols. + SymbolId nextSymbol{ 1U }; Prepared::BlockId nextBlock{ 1U }; - SymbolId nextFunction{ 1U }; std::vector pendingFunctions; // Innermost first; every pair stores (break exit, continue target). std::vector> loopTargets; }; - struct Atomized final - { - std::vector prefix; - Prepared::Atom atom; - State state; - }; - - struct OperationResult final - { - std::vector prefix; - Prepared::Operation operation{ Prepared::Operation::Copy }; - std::vector operands; - SymbolName closureFunction{}; - std::vector captures; - State state; - }; - - struct BlocksResult final - { - std::vector blocks; - State state; - }; - [[nodiscard]] auto LowerLiteral(const Expression &expression) -> Prepared::Atom { @@ -118,27 +100,95 @@ namespace Visual::XSharp::Core::CorePrep std::abort(); } + /** + * @brief The block receiving instructions plus every finished block. + * + * Expression atomization is not confined to one block: a + * short-circuit operator ends the current block with a branch and + * continues in a join block. Threading one cursor through both + * expressions and statements lets either of them close and reopen + * blocks without a second lowering path. + */ + struct Cursor final + { + State state; + Prepared::BlockId block{}; + std::vector instructions; + std::vector closed; + /// False after a return, break or continue ended the region. + bool open{ true }; + + void + Emit(Prepared::Instruction instruction) + { + instructions.push_back(std::move(instruction)); + } + + /// End the current block; no block is open until Open is called. + void + Close(Prepared::Terminator terminator) + { + closed.push_back( + { block, std::move(instructions), std::move(terminator) }); + instructions.clear(); + open = false; + } + + void + Open(const Prepared::BlockId id) + { + block = id; + instructions.clear(); + open = true; + } + + void + Jump(const Prepared::BlockId target) + { + Close({ Prepared::Terminator::Kind::Jump, {}, target, 0U }); + } + + void + Branch(Prepared::Atom condition, + const Prepared::BlockId whenTrue, + const Prepared::BlockId whenFalse) + { + Close({ Prepared::Terminator::Kind::Branch, + std::move(condition), + whenTrue, + whenFalse }); + } + + [[nodiscard]] auto + Temporary(std::u32string prefix) -> SymbolName + { + const auto id = state.nextSymbol++; + const auto digits = std::to_string(id); + prefix.append(digits.begin(), digits.end()); + return SymbolName{ id, std::move(prefix) }; + } + }; + + struct OperationResult final + { + Prepared::Operation operation{ Prepared::Operation::Copy }; + std::vector operands; + SymbolName closureFunction{}; + std::vector captures; + }; + [[nodiscard]] auto - Atomize(State state, const Expression &expression) -> Atomized; + Atomize(Cursor &cursor, const Expression &expression) -> Prepared::Atom; [[nodiscard]] auto - AtomizeMany(State state, const std::vector &expressions) - -> std::pair, - std::pair, State>> + AtomizeMany(Cursor &cursor, const std::vector &expressions) + -> std::vector { - std::vector prefix; std::vector atoms; atoms.reserve(expressions.size()); for (const auto &expression : expressions) - { - auto atomized = Atomize(state, expression); - prefix.insert(prefix.end(), - std::make_move_iterator(atomized.prefix.begin()), - std::make_move_iterator(atomized.prefix.end())); - atoms.push_back(std::move(atomized.atom)); - state = atomized.state; - } - return { std::move(prefix), { std::move(atoms), state } }; + atoms.push_back(Atomize(cursor, expression)); + return atoms; } [[nodiscard]] auto @@ -152,20 +202,18 @@ namespace Visual::XSharp::Core::CorePrep type); } + /// Canonicalize a numeric truth value to `value != 0` in the open + /// block; a Boolean atom is returned unchanged. [[nodiscard]] auto - Booleanize(State state, Prepared::Atom atom) -> Atomized + Booleanize(Cursor &cursor, Prepared::Atom atom) -> Prepared::Atom { if (atom.type == Type::boolean()) - return { {}, std::move(atom), std::move(state) }; - const auto id = state.nextTemporary++; - const auto digits = std::to_string(id); - std::u32string spelling = U"$condition"; - spelling.append(digits.begin(), digits.end()); - auto temporary = SymbolName{ id, std::move(spelling) }; + return atom; + auto temporary = cursor.Temporary(U"$condition"); std::vector operands; operands.push_back(std::move(atom)); operands.push_back(ZeroForBooleanContext(operands.front().type)); - Prepared::Instruction comparison{ + cursor.Emit(Prepared::Instruction{ Prepared::Instruction::Kind::Bind, temporary, Type::boolean(), @@ -174,47 +222,84 @@ namespace Visual::XSharp::Core::CorePrep std::move(operands), {}, {}, - }; - return { - { std::move(comparison) }, - Prepared::Atom::variable(std::move(temporary), Type::boolean()), - std::move(state), - }; + }); + return Prepared::Atom::variable(std::move(temporary), + Type::boolean()); } [[nodiscard]] auto - BooleanizeMany(State state, std::vector atoms) - -> std::pair, - std::pair, State>> + IsShortCircuit(const Expression &expression) -> bool { - std::vector prefix; - std::vector booleans; - booleans.reserve(atoms.size()); - for (auto &atom : atoms) - { - auto boolean = Booleanize(std::move(state), std::move(atom)); - prefix.insert(prefix.end(), - std::make_move_iterator(boolean.prefix.begin()), - std::make_move_iterator(boolean.prefix.end())); - booleans.push_back(std::move(boolean.atom)); - state = std::move(boolean.state); - } - return { std::move(prefix), - { std::move(booleans), std::move(state) } }; + return expression.kind == Expression::Kind::Primitive + && (expression.primitive == Primitive::LogicalAnd + || expression.primitive == Primitive::LogicalOr) + && expression.operands.size() == 2U; } - struct CapturesResult final + /** + * @brief Lower `&&` and `||` as control flow, not as an eager operator. + * + * The right operand is evaluated only on the path that needs it, so + * its calls, traps and non-termination stay conditional exactly as + * the source wrote them. The result slot is initialized with the + * short-circuit value before the branch and overwritten only on the + * path that evaluates the right operand; every predecessor of the + * join therefore carries an initialized Boolean without a phi node + * in CorePrep's storage-oriented form. This matches the Haskell + * CorePrep lowering. + */ + [[nodiscard]] auto + AtomizeShortCircuit(Cursor &cursor, const Expression &expression) + -> Prepared::Atom { - std::vector prefix; - std::vector captures; - State state; - }; + const auto isOr = expression.primitive == Primitive::LogicalOr; + auto condition + = Booleanize(cursor, + Atomize(cursor, expression.operands.front())); + auto result = cursor.Temporary(U"$shortcircuit"); + cursor.Emit(Prepared::Instruction{ + Prepared::Instruction::Kind::Bind, + result, + Type::boolean(), + true, + Prepared::Operation::Copy, + { Prepared::Atom::constant(Prepared::Literal{ isOr }, + Type::boolean()) }, + {}, + {}, + }); + const auto rightId = cursor.state.nextBlock; + const auto joinId = rightId + 1U; + cursor.state.nextBlock = joinId + 1U; + if (isOr) + cursor.Branch(std::move(condition), joinId, rightId); + else + cursor.Branch(std::move(condition), rightId, joinId); + + cursor.Open(rightId); + auto right + = Booleanize(cursor, + Atomize(cursor, expression.operands.back())); + cursor.Emit(Prepared::Instruction{ + Prepared::Instruction::Kind::Assign, + result, + Type::boolean(), + false, + Prepared::Operation::Copy, + { std::move(right) }, + {}, + {}, + }); + cursor.Jump(joinId); + cursor.Open(joinId); + return Prepared::Atom::variable(std::move(result), Type::boolean()); + } struct PreparedFunctionResult final { Prepared::Function function; std::vector pendingFunctions; - SymbolId nextFunction{}; + SymbolId nextSymbol{}; }; [[nodiscard]] auto @@ -304,10 +389,9 @@ namespace Visual::XSharp::Core::CorePrep } [[nodiscard]] auto - AtomizeCaptures(State state, const std::vector &captures) - -> CapturesResult + AtomizeCaptures(Cursor &cursor, const std::vector &captures) + -> std::vector { - std::vector prefix; std::vector prepared; prepared.reserve(captures.size()); for (const auto &capture : captures) @@ -316,84 +400,60 @@ namespace Visual::XSharp::Core::CorePrep // for direct native API callers as well. if (!capture.value) std::abort(); - auto value = Atomize(state, *capture.value); - prefix.insert(prefix.end(), - std::make_move_iterator(value.prefix.begin()), - std::make_move_iterator(value.prefix.end())); prepared.push_back(Prepared::Capture{ capture.mode, capture.symbol, capture.type, - std::move(value.atom), + Atomize(cursor, *capture.value), }); - state = std::move(value.state); } - return { std::move(prefix), std::move(prepared), std::move(state) }; + return prepared; } [[nodiscard]] auto - AtomizeOperation(State state, const Expression &expression) + AtomizeOperation(Cursor &cursor, const Expression &expression) -> OperationResult { - if (expression.kind == Expression::Kind::Let) - { - auto atomized = Atomize(std::move(state), expression); - return { std::move(atomized.prefix), - Prepared::Operation::Copy, - { std::move(atomized.atom) }, + if (expression.kind == Expression::Kind::Let + || IsShortCircuit(expression)) + return { Prepared::Operation::Copy, + { Atomize(cursor, expression) }, {}, - {}, - std::move(atomized.state) }; - } + {} }; if (expression.kind == Expression::Kind::Variable) - return { {}, - Prepared::Operation::Copy, + return { Prepared::Operation::Copy, { Prepared::Atom::variable(expression.symbol, expression.type) }, {}, - {}, - state }; + {} }; if (expression.kind == Expression::Kind::Literal) - return { {}, - Prepared::Operation::Copy, + return { Prepared::Operation::Copy, { LowerLiteral(expression) }, {}, - {}, - state }; + {} }; if (expression.kind == Expression::Kind::Apply) { - auto callee = Atomize(state, *expression.callee); - auto arguments = AtomizeMany(callee.state, expression.operands); - callee.prefix.insert( - callee.prefix.end(), - std::make_move_iterator(arguments.first.begin()), - std::make_move_iterator(arguments.first.end())); std::vector operands; - operands.reserve(arguments.second.first.size() + 1U); - operands.push_back(std::move(callee.atom)); - operands.insert( - operands.end(), - std::make_move_iterator(arguments.second.first.begin()), - std::make_move_iterator(arguments.second.first.end())); - return { std::move(callee.prefix), - Prepared::Operation::Call, + operands.reserve(expression.operands.size() + 1U); + operands.push_back(Atomize(cursor, *expression.callee)); + for (const auto &argument : expression.operands) + operands.push_back(Atomize(cursor, argument)); + return { Prepared::Operation::Call, std::move(operands), {}, - {}, - arguments.second.second }; + {} }; } if (expression.kind == Expression::Kind::Closure) { if (!expression.closureBody) std::abort(); - const auto closureId = state.nextFunction++; + const auto closureId = cursor.state.nextSymbol++; const auto digits = std::to_string(closureId); std::u32string spelling = U"$closure"; spelling.append(digits.begin(), digits.end()); SymbolName closureName{ closureId, std::move(spelling) }; - auto captures - = AtomizeCaptures(std::move(state), expression.captures); + auto captures = AtomizeCaptures(cursor, expression.captures); std::vector parameters; parameters.reserve(expression.captures.size() + expression.closureParameters.size()); @@ -402,520 +462,349 @@ namespace Visual::XSharp::Core::CorePrep Parameter{ capture.symbol, capture.type }); for (const auto &[symbol, type] : expression.closureParameters) parameters.push_back(Parameter{ symbol, type }); - captures.state.pendingFunctions.push_back(Function{ + cursor.state.pendingFunctions.push_back(Function{ closureName, std::move(parameters), expression.closureReturnType, *expression.closureBody, }); - return { - std::move(captures.prefix), - Prepared::Operation::MakeClosure, - {}, - std::move(closureName), - std::move(captures.captures), - std::move(captures.state), - }; + return { Prepared::Operation::MakeClosure, + {}, + std::move(closureName), + std::move(captures) }; } - auto arguments = AtomizeMany(state, expression.operands); + auto operands = AtomizeMany(cursor, expression.operands); if (expression.primitive == Primitive::LogicalAnd || expression.primitive == Primitive::LogicalOr || expression.primitive == Primitive::LogicalNot) - { - auto booleans - = BooleanizeMany(std::move(arguments.second.second), - std::move(arguments.second.first)); - arguments.first.insert( - arguments.first.end(), - std::make_move_iterator(booleans.first.begin()), - std::make_move_iterator(booleans.first.end())); - return { std::move(arguments.first), - LowerPrimitive(expression.primitive), - std::move(booleans.second.first), - {}, - {}, - std::move(booleans.second.second) }; - } - return { std::move(arguments.first), - LowerPrimitive(expression.primitive), - std::move(arguments.second.first), - {}, + for (auto &operand : operands) + operand = Booleanize(cursor, std::move(operand)); + return { LowerPrimitive(expression.primitive), + std::move(operands), {}, - arguments.second.second }; + {} }; } [[nodiscard]] auto - Atomize(State state, const Expression &expression) -> Atomized + Atomize(Cursor &cursor, const Expression &expression) -> Prepared::Atom { if (expression.kind == Expression::Kind::Variable) - return { {}, - Prepared::Atom::variable(expression.symbol, - expression.type), - state }; + return Prepared::Atom::variable(expression.symbol, + expression.type); if (expression.kind == Expression::Kind::Literal) - return { {}, LowerLiteral(expression), state }; + return LowerLiteral(expression); if (expression.kind == Expression::Kind::Let) { if (!expression.letValue || !expression.letBody) std::abort(); - auto value = Atomize(std::move(state), *expression.letValue); - value.prefix.push_back( + auto value = Atomize(cursor, *expression.letValue); + cursor.Emit( Prepared::Instruction{ Prepared::Instruction::Kind::Bind, expression.letSymbol, expression.letType, false, Prepared::Operation::Copy, - { std::move(value.atom) }, + { std::move(value) }, {}, {} }); - auto body - = Atomize(std::move(value.state), *expression.letBody); - value.prefix.insert( - value.prefix.end(), - std::make_move_iterator(body.prefix.begin()), - std::make_move_iterator(body.prefix.end())); - return { std::move(value.prefix), - std::move(body.atom), - std::move(body.state) }; + return Atomize(cursor, *expression.letBody); } - - auto operation = AtomizeOperation(state, expression); - const auto id = operation.state.nextTemporary++; - const auto digits = std::to_string(id); - std::u32string spelling = U"$coreprep"; - spelling.append(digits.begin(), digits.end()); - auto temporary = SymbolName{ id, std::move(spelling) }; - Prepared::Instruction binding{ Prepared::Instruction::Kind::Bind, - temporary, - expression.type, - false, - operation.operation, - std::move(operation.operands), - std::move(operation.closureFunction), - std::move(operation.captures) }; - operation.prefix.push_back(std::move(binding)); - return { std::move(operation.prefix), - Prepared::Atom::variable(std::move(temporary), - expression.type), - operation.state }; + if (IsShortCircuit(expression)) + return AtomizeShortCircuit(cursor, expression); + + auto operation = AtomizeOperation(cursor, expression); + auto temporary = cursor.Temporary(U"$coreprep"); + cursor.Emit( + Prepared::Instruction{ Prepared::Instruction::Kind::Bind, + temporary, + expression.type, + false, + operation.operation, + std::move(operation.operands), + std::move(operation.closureFunction), + std::move(operation.captures) }); + return Prepared::Atom::variable(std::move(temporary), + expression.type); } - [[nodiscard]] auto - PrepareStatements(State state, - Prepared::BlockId blockId, - std::vector instructions, - const std::vector &statements, - std::size_t start = 0U) -> BlocksResult; - - /** - * @brief Replace open-region fallthrough sentinels with a real edge. - * - * `PrepareStatements` uses `Unreachable` for the still-open tail of a - * nested region. Explicit `Return` and `Jump` terminators are closed - * paths and are deliberately not rewritten. - */ void - ConnectFallthrough(std::vector &blocks, - const Prepared::BlockId target) - { - for (auto &block : blocks) - { - if (block.terminator.kind - == Prepared::Terminator::Kind::Unreachable) - { - block.terminator.kind = Prepared::Terminator::Kind::Jump; - block.terminator.true_target = target; - } - } - } + PrepareStatements(Cursor &cursor, + const std::vector &statements); /** - * @brief Prepare a loop region with a scoped transfer-target stack. + * @brief Prepare one loop region with a scoped transfer-target pair. + * + * The region starts in a fresh block. While it is built, `break` + * jumps to breakTarget and `continue` to continueTarget; the + * enclosing pair is restored afterwards, so an inner transfer + * cannot target an outer loop. A region that is still open at its + * end jumps to fallthroughTarget. * - * Nested regions inherit this loop's targets while they are built, - * then restore the enclosing stack. Thus an inner `break` or - * `continue` cannot accidentally target an outer loop. Only blocks - * still marked as fallthrough are connected to the continuation; - * explicit return and jump terminators remain untouched. + * The fallthrough successor is a separate argument because it is + * not always the `continue` target: a for-loop update region is + * entered by `continue` but must fall through to the condition. + * Reusing the continue target there makes the region branch to + * itself and never re-test the loop condition. */ - [[nodiscard]] auto - PrepareLoopRegion(State state, + void + PrepareLoopRegion(Cursor &cursor, const Prepared::BlockId startId, const Prepared::BlockId breakTarget, const Prepared::BlockId continueTarget, + const Prepared::BlockId fallthroughTarget, const std::vector &statements) - -> BlocksResult { - auto enclosingTargets = state.loopTargets; - state.loopTargets.emplace_back(breakTarget, continueTarget); - auto result = PrepareStatements(state, startId, {}, statements); - ConnectFallthrough(result.blocks, continueTarget); - result.state.loopTargets = std::move(enclosingTargets); - return result; + cursor.state.loopTargets.emplace_back(breakTarget, continueTarget); + cursor.Open(startId); + PrepareStatements(cursor, statements); + if (cursor.open) + cursor.Jump(fallthroughTarget); + cursor.state.loopTargets.pop_back(); } + /// Evaluate a loop or branch condition in the open block and return + /// its Boolean atom. Short-circuit operands may leave a different + /// block open than the one the condition started in. [[nodiscard]] auto - PrepareBranch(State state, - Prepared::BlockId blockId, - Prepared::BlockId joinId, - const std::vector &statements) -> BlocksResult + PrepareCondition(Cursor &cursor, const Expression &condition) + -> Prepared::Atom { - auto result = PrepareStatements(state, blockId, {}, statements); - for (auto &block : result.blocks) - { - if (block.terminator.kind - == Prepared::Terminator::Kind::Unreachable) - { - block.terminator.kind = Prepared::Terminator::Kind::Jump; - block.terminator.true_target = joinId; - } - } - return result; + return Booleanize(cursor, Atomize(cursor, condition)); } - [[nodiscard]] auto - PrepareStatements(State state, - Prepared::BlockId blockId, - std::vector instructions, - const std::vector &statements, - std::size_t start) -> BlocksResult + void + PrepareBranchRegion(Cursor &cursor, + const Prepared::BlockId startId, + const Prepared::BlockId joinId, + const std::vector &statements) { - for (std::size_t index = start; index < statements.size(); ++index) - { - const auto &statement = statements[index]; - if (statement.kind == Statement::Kind::Bind) - { - auto operation - = AtomizeOperation(state, statement.binding.value); - instructions.insert( - instructions.end(), - std::make_move_iterator(operation.prefix.begin()), - std::make_move_iterator(operation.prefix.end())); - instructions.push_back(Prepared::Instruction{ - Prepared::Instruction::Kind::Bind, - statement.binding.symbol, - statement.binding.type, - statement.binding.mutableBinding, - operation.operation, - std::move(operation.operands), - std::move(operation.closureFunction), - std::move(operation.captures) }); - state = operation.state; - continue; - } - if (statement.kind == Statement::Kind::Assign) - { - auto value = Atomize(state, statement.expression); - instructions.insert( - instructions.end(), - std::make_move_iterator(value.prefix.begin()), - std::make_move_iterator(value.prefix.end())); - instructions.push_back(Prepared::Instruction{ - Prepared::Instruction::Kind::Assign, - statement.destination, - statement.expression.type, - false, - Prepared::Operation::Copy, - { std::move(value.atom) }, - {}, - {} }); - state = value.state; - continue; - } - if (statement.kind == Statement::Kind::Evaluate) - { - auto operation - = AtomizeOperation(state, statement.expression); - instructions.insert( - instructions.end(), - std::make_move_iterator(operation.prefix.begin()), - std::make_move_iterator(operation.prefix.end())); - instructions.push_back(Prepared::Instruction{ - Prepared::Instruction::Kind::Evaluate, - {}, - statement.expression.type, - false, - operation.operation, - std::move(operation.operands), - std::move(operation.closureFunction), - std::move(operation.captures) }); - state = operation.state; - continue; - } - if (statement.kind == Statement::Kind::Return) - { - auto value = Atomize(state, statement.expression); - instructions.insert( - instructions.end(), - std::make_move_iterator(value.prefix.begin()), - std::make_move_iterator(value.prefix.end())); - return { { { blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Return, - std::move(value.atom), - 0U, - 0U } } }, - value.state }; - } + cursor.Open(startId); + PrepareStatements(cursor, statements); + if (cursor.open) + cursor.Jump(joinId); + } - if (statement.kind == Statement::Kind::Break - || statement.kind == Statement::Kind::Continue) + /** + * @brief Append structured statements to the cursor's open block. + * + * On return the cursor is either still open, meaning control falls + * through to whatever the caller places next, or closed by a + * return, break or continue, after which the remaining statements + * of the region are unreachable and are not lowered. + */ + void + PrepareStatements(Cursor &cursor, + const std::vector &statements) + { + for (const auto &statement : statements) + { + switch (statement.kind) { - if (state.loopTargets.empty()) + case Statement::Kind::Bind: + { + auto operation + = AtomizeOperation(cursor, statement.binding.value); + cursor.Emit(Prepared::Instruction{ + Prepared::Instruction::Kind::Bind, + statement.binding.symbol, + statement.binding.type, + statement.binding.mutableBinding, + operation.operation, + std::move(operation.operands), + std::move(operation.closureFunction), + std::move(operation.captures) }); + break; + } + case Statement::Kind::Assign: { - return { { { blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Unreachable, - {}, + auto value = Atomize(cursor, statement.expression); + cursor.Emit(Prepared::Instruction{ + Prepared::Instruction::Kind::Assign, + statement.destination, + statement.expression.type, + false, + Prepared::Operation::Copy, + { std::move(value) }, + {}, + {} }); + break; + } + case Statement::Kind::Evaluate: + { + auto operation + = AtomizeOperation(cursor, statement.expression); + cursor.Emit(Prepared::Instruction{ + Prepared::Instruction::Kind::Evaluate, + {}, + statement.expression.type, + false, + operation.operation, + std::move(operation.operands), + std::move(operation.closureFunction), + std::move(operation.captures) }); + break; + } + case Statement::Kind::Return: + { + auto value = Atomize(cursor, statement.expression); + cursor.Close({ Prepared::Terminator::Kind::Return, + std::move(value), 0U, - 0U } } }, - state }; + 0U }); + return; + } + case Statement::Kind::Break: + case Statement::Kind::Continue: + { + // Core verification rejects a transfer outside a + // loop; stay total for direct native API callers. + if (cursor.state.loopTargets.empty()) + { + cursor.Close( + { Prepared::Terminator::Kind::Unreachable, + {}, + 0U, + 0U }); + return; + } + const auto targets = cursor.state.loopTargets.back(); + cursor.Jump(statement.kind == Statement::Kind::Break + ? targets.first + : targets.second); + return; + } + case Statement::Kind::While: + { + // The condition owns a dedicated header block. + // Folding it into the incoming block would make the + // back-edge re-execute every straight-line statement + // that precedes the loop, including the initializers + // it tests. + const auto conditionId = cursor.state.nextBlock; + const auto bodyId = conditionId + 1U; + const auto exitId = bodyId + 1U; + cursor.state.nextBlock = exitId + 1U; + + cursor.Jump(conditionId); + cursor.Open(conditionId); + cursor.Branch( + PrepareCondition(cursor, statement.expression), + bodyId, + exitId); + PrepareLoopRegion(cursor, + bodyId, + exitId, + conditionId, + conditionId, + statement.loopBody); + cursor.Open(exitId); + break; + } + case Statement::Kind::DoWhile: + { + const auto bodyId = cursor.state.nextBlock; + const auto conditionId = bodyId + 1U; + const auto exitId = conditionId + 1U; + cursor.state.nextBlock = exitId + 1U; + + cursor.Jump(bodyId); + PrepareLoopRegion(cursor, + bodyId, + exitId, + conditionId, + conditionId, + statement.loopBody); + cursor.Open(conditionId); + cursor.Branch( + PrepareCondition(cursor, statement.expression), + bodyId, + exitId); + cursor.Open(exitId); + break; + } + case Statement::Kind::For: + { + const auto conditionId = cursor.state.nextBlock; + const auto bodyId = conditionId + 1U; + const auto updateId = bodyId + 1U; + const auto exitId = updateId + 1U; + cursor.state.nextBlock = exitId + 1U; + + cursor.Jump(conditionId); + cursor.Open(conditionId); + // A numeric condition's canonicalizing comparison is + // part of the header and is re-evaluated each pass. + cursor.Branch( + PrepareCondition(cursor, statement.expression), + bodyId, + exitId); + PrepareLoopRegion(cursor, + bodyId, + exitId, + updateId, + updateId, + statement.loopBody); + // `continue` enters the update region; the region + // itself then returns to the condition, never to its + // own entry. + PrepareLoopRegion(cursor, + updateId, + exitId, + updateId, + conditionId, + statement.loopUpdate); + cursor.Open(exitId); + break; + } + case Statement::Kind::If: + { + auto condition + = PrepareCondition(cursor, statement.expression); + const auto trueId = cursor.state.nextBlock; + const auto falseId = trueId + 1U; + const auto joinId = falseId + 1U; + cursor.state.nextBlock = joinId + 1U; + cursor.Branch(std::move(condition), trueId, falseId); + PrepareBranchRegion(cursor, + trueId, + joinId, + statement.trueBranch); + PrepareBranchRegion(cursor, + falseId, + joinId, + statement.falseBranch); + cursor.Open(joinId); + break; } - const auto targets = state.loopTargets.back(); - const auto target = statement.kind == Statement::Kind::Break - ? targets.first - : targets.second; - return { { { blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Jump, - {}, - target, - 0U } } }, - state }; - } - - if (statement.kind == Statement::Kind::While) - { - auto condition = Atomize(state, statement.expression); - auto boolean = Booleanize(std::move(condition.state), - std::move(condition.atom)); - condition.prefix.insert( - condition.prefix.end(), - std::make_move_iterator(boolean.prefix.begin()), - std::make_move_iterator(boolean.prefix.end())); - instructions.insert( - instructions.end(), - std::make_move_iterator(condition.prefix.begin()), - std::make_move_iterator(condition.prefix.end())); - - const auto bodyId = boolean.state.nextBlock; - const auto exitId = bodyId + 1U; - boolean.state.nextBlock = exitId + 1U; - auto body = PrepareLoopRegion(boolean.state, - bodyId, - exitId, - blockId, - statement.loopBody); - auto tail = PrepareStatements(body.state, - exitId, - {}, - statements, - index + 1U); - std::vector blocks; - blocks.push_back({ blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Branch, - std::move(boolean.atom), - bodyId, - exitId } }); - blocks.insert(blocks.end(), - std::make_move_iterator(body.blocks.begin()), - std::make_move_iterator(body.blocks.end())); - blocks.insert(blocks.end(), - std::make_move_iterator(tail.blocks.begin()), - std::make_move_iterator(tail.blocks.end())); - return { std::move(blocks), std::move(tail.state) }; - } - - if (statement.kind == Statement::Kind::DoWhile) - { - const auto bodyId = state.nextBlock; - const auto conditionId = bodyId + 1U; - const auto exitId = conditionId + 1U; - state.nextBlock = exitId + 1U; - auto body = PrepareLoopRegion(state, - bodyId, - exitId, - conditionId, - statement.loopBody); - auto condition = Atomize(body.state, statement.expression); - auto boolean = Booleanize(std::move(condition.state), - std::move(condition.atom)); - condition.prefix.insert( - condition.prefix.end(), - std::make_move_iterator(boolean.prefix.begin()), - std::make_move_iterator(boolean.prefix.end())); - auto tail = PrepareStatements(boolean.state, - exitId, - {}, - statements, - index + 1U); - - std::vector blocks; - blocks.push_back({ blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Jump, - {}, - bodyId, - 0U } }); - blocks.insert(blocks.end(), - std::make_move_iterator(body.blocks.begin()), - std::make_move_iterator(body.blocks.end())); - blocks.push_back({ conditionId, - std::move(condition.prefix), - { Prepared::Terminator::Kind::Branch, - std::move(boolean.atom), - bodyId, - exitId } }); - blocks.insert(blocks.end(), - std::make_move_iterator(tail.blocks.begin()), - std::make_move_iterator(tail.blocks.end())); - return { std::move(blocks), std::move(tail.state) }; - } - - if (statement.kind == Statement::Kind::For) - { - const auto conditionId = state.nextBlock; - const auto bodyId = conditionId + 1U; - const auto updateId = bodyId + 1U; - const auto exitId = updateId + 1U; - state.nextBlock = exitId + 1U; - - auto condition = Atomize(state, statement.expression); - auto boolean = Booleanize(std::move(condition.state), - std::move(condition.atom)); - auto body = PrepareLoopRegion(boolean.state, - bodyId, - exitId, - updateId, - statement.loopBody); - auto update = PrepareLoopRegion(body.state, - updateId, - exitId, - updateId, - statement.loopUpdate); - auto tail = PrepareStatements(update.state, - exitId, - {}, - statements, - index + 1U); - - std::vector blocks; - blocks.push_back({ blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Jump, - {}, - conditionId, - 0U } }); - blocks.push_back({ conditionId, - std::move(condition.prefix), - { Prepared::Terminator::Kind::Branch, - std::move(boolean.atom), - bodyId, - exitId } }); - blocks.insert(blocks.end(), - std::make_move_iterator(body.blocks.begin()), - std::make_move_iterator(body.blocks.end())); - blocks.insert( - blocks.end(), - std::make_move_iterator(update.blocks.begin()), - std::make_move_iterator(update.blocks.end())); - blocks.insert(blocks.end(), - std::make_move_iterator(tail.blocks.begin()), - std::make_move_iterator(tail.blocks.end())); - return { std::move(blocks), std::move(tail.state) }; - } - - if (statement.kind != Statement::Kind::If) - { - return { { { blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Unreachable, - {}, - 0U, - 0U } } }, - state }; } - auto condition = Atomize(state, statement.expression); - auto boolean = Booleanize(std::move(condition.state), - std::move(condition.atom)); - instructions.insert( - instructions.end(), - std::make_move_iterator(condition.prefix.begin()), - std::make_move_iterator(condition.prefix.end())); - instructions.insert( - instructions.end(), - std::make_move_iterator(boolean.prefix.begin()), - std::make_move_iterator(boolean.prefix.end())); - const auto trueId = boolean.state.nextBlock; - const auto falseId = trueId + 1U; - const auto joinId = falseId + 1U; - boolean.state.nextBlock = joinId + 1U; - auto trueBlocks = PrepareBranch(boolean.state, - trueId, - joinId, - statement.trueBranch); - auto falseBlocks = PrepareBranch(trueBlocks.state, - falseId, - joinId, - statement.falseBranch); - auto tail = PrepareStatements(falseBlocks.state, - joinId, - {}, - statements, - index + 1U); - std::vector blocks; - blocks.push_back({ blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Branch, - std::move(boolean.atom), - trueId, - falseId } }); - blocks.insert( - blocks.end(), - std::make_move_iterator(trueBlocks.blocks.begin()), - std::make_move_iterator(trueBlocks.blocks.end())); - blocks.insert( - blocks.end(), - std::make_move_iterator(falseBlocks.blocks.begin()), - std::make_move_iterator(falseBlocks.blocks.end())); - blocks.insert(blocks.end(), - std::make_move_iterator(tail.blocks.begin()), - std::make_move_iterator(tail.blocks.end())); - return { std::move(blocks), tail.state }; } - return { - { { blockId, - std::move(instructions), - { Prepared::Terminator::Kind::Unreachable, {}, 0U, 0U } } }, - state - }; } [[nodiscard]] auto - PrepareFunction(const Function &function, SymbolId nextFunction) + PrepareFunction(const Function &function, SymbolId nextSymbol) -> PreparedFunctionResult { - SymbolId highest = HighestFunctionSymbol(function); std::vector parameters; parameters.reserve(function.parameters.size()); for (const auto ¶meter : function.parameters) { parameters.push_back({ parameter.symbol, parameter.type }); } - auto body = PrepareStatements( - State{ highest + 1U, 1U, nextFunction, {}, {} }, - 0U, - {}, - function.body); + Cursor cursor{ State{ nextSymbol, 1U, {}, {} }, 0U, {}, {}, true }; + PrepareStatements(cursor, function.body); + // Core verification proves every path returns. A body that still + // falls off its end is marked instead of given an invented value. + if (cursor.open) + cursor.Close( + { Prepared::Terminator::Kind::Unreachable, {}, 0U, 0U }); return { Prepared::Function{ function.symbol, std::move(parameters), function.returnType, 0U, - std::move(body.blocks) }, - std::move(body.state.pendingFunctions), - body.state.nextFunction, + std::move(cursor.closed) }, + std::move(cursor.state.pendingFunctions), + cursor.state.nextSymbol, }; } } // namespace @@ -930,20 +819,23 @@ namespace Visual::XSharp::Core::CorePrep pending.reserve(module.functions.size()); for (const auto &function : module.functions) pending.emplace_back(function, function.sourceFile); - SymbolId nextFunction = 1U; + // Lifted closure bodies only mention identities already counted + // here or generated by the shared counter, so the seed stays valid + // for functions appended to the queue later. + SymbolId nextSymbol = 1U; for (const auto &[function, _] : pending) - nextFunction - = std::max(nextFunction, HighestFunctionSymbol(function) + 1U); + nextSymbol + = std::max(nextSymbol, HighestFunctionSymbol(function) + 1U); // Closure bodies are lifted as ordinary Core functions and fed back // through the same work queue. This naturally handles nested closures // without adding a second, subtly different lowering implementation. for (std::size_t index = 0U; index < pending.size(); ++index) { - auto result = PrepareFunction(pending[index].first, nextFunction); + auto result = PrepareFunction(pending[index].first, nextSymbol); result.function.sourceFile = pending[index].second; prepared.functions.push_back(std::move(result.function)); - nextFunction = result.nextFunction; + nextSymbol = result.nextSymbol; for (auto &function : result.pendingFunctions) pending.emplace_back(std::move(function), pending[index].second); diff --git a/Compiler/Core/Tests/BUILD.bazel b/Compiler/Core/Tests/BUILD.bazel index 177c4bc8..50b1cc03 100644 --- a/Compiler/Core/Tests/BUILD.bazel +++ b/Compiler/Core/Tests/BUILD.bazel @@ -16,6 +16,9 @@ cc_binary( name = "core_pipeline_tests", srcs = [ "CorePipelineTests.cpp", + "LoopLoweringTests.cpp", + "ShortCircuitLoweringTests.cpp", + "SymbolAllocationTests.cpp", "TemplateTests.cpp", ], deps = [ diff --git a/Compiler/Core/Tests/CorePipelineTests.cpp b/Compiler/Core/Tests/CorePipelineTests.cpp index 89693ffd..e6c7edcc 100644 --- a/Compiler/Core/Tests/CorePipelineTests.cpp +++ b/Compiler/Core/Tests/CorePipelineTests.cpp @@ -681,15 +681,37 @@ TEST_CASE( REQUIRE(Core::Verify(module).empty()); const auto prepared = Core::CorePrep::Prepare(module); REQUIRE(visual_xsharp::core::verify(prepared).empty()); - const auto &instructions - = prepared.functions.front().blocks.front().instructions; - REQUIRE(instructions.size() == 3U); - CHECK(instructions.at(0).operation + // Each operand is canonicalized with its own typed `!= 0` comparison. + // The conjunction itself is control flow: the left comparison and the + // result slot live in the entry block, and the right comparison runs + // only in the block the true edge reaches. + const auto &blocks = prepared.functions.front().blocks; + REQUIRE(blocks.size() == 3U); + const auto &entry = blocks.front(); + REQUIRE(entry.instructions.size() == 2U); + CHECK(entry.instructions.at(0).operation == visual_xsharp::core::Operation::NotEqual); - CHECK(instructions.at(1).operation + CHECK(entry.instructions.at(0).operands.front().type + == Core::Type::int64()); + CHECK(entry.instructions.at(1).operation + == visual_xsharp::core::Operation::Copy); + REQUIRE(entry.terminator.kind + == visual_xsharp::core::Terminator::Kind::Branch); + const auto right = std::ranges::find(blocks, + entry.terminator.true_target, + &visual_xsharp::core::Block::id); + REQUIRE(right != blocks.end()); + REQUIRE(right->instructions.size() == 2U); + CHECK(right->instructions.at(0).operation == visual_xsharp::core::Operation::NotEqual); - CHECK(instructions.at(2).operation - == visual_xsharp::core::Operation::LogicalAnd); + CHECK(right->instructions.at(0).operands.front().type + == Core::Type::float32()); + CHECK(right->instructions.at(1).kind + == visual_xsharp::core::Instruction::Kind::Assign); + for (const auto &block : blocks) + for (const auto &instruction : block.instructions) + CHECK(instruction.operation + != visual_xsharp::core::Operation::LogicalAnd); const auto result = Visual::XSharp::Pipeline::ConsumeCore( Core::Wire::Encode(module).bytes); REQUIRE(result); diff --git a/Compiler/Core/Tests/LoopLoweringTests.cpp b/Compiler/Core/Tests/LoopLoweringTests.cpp new file mode 100644 index 00000000..cd84a2cd --- /dev/null +++ b/Compiler/Core/Tests/LoopLoweringTests.cpp @@ -0,0 +1,453 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Core/CorePrep/Prepare.hpp" +#include "Visual/XSharp/Core/CorePrep/Verifier.hpp" +#include "Visual/XSharp/Core/Verifier.hpp" + +// These tests pin the control-flow graph the native Core-to-CorePrep adapter +// builds for structured loops. They assert exact edges instead of "some +// back-edge exists": a for-loop whose update region branched to itself used +// to satisfy every weaker shape check while never terminating. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Prepared = visual_xsharp::core; + + constexpr std::uint64_t kTotal = 2U; + constexpr std::uint64_t kIndex = 3U; + + [[nodiscard]] auto + Integer(std::int64_t value) -> Core::Expression + { + return Core::Expression::Constant(value, Core::Type::int64()); + } + + [[nodiscard]] auto + Spelling(std::uint64_t id) -> std::u32string + { + return id == kTotal ? U"total" : id == kIndex ? U"index" : U"inner"; + } + + [[nodiscard]] auto + Variable(std::uint64_t id, std::u32string spelling) -> Core::Expression + { + return Core::Expression::Variable({ id, std::move(spelling) }, + Core::Type::int64()); + } + + [[nodiscard]] auto + Compare(Core::Primitive operation, std::uint64_t id, std::int64_t value) + -> Core::Expression + { + return Core::Expression::InvokePrimitive( + operation, + { Variable(id, Spelling(id)), Integer(value) }, + Core::Type::boolean()); + } + + [[nodiscard]] auto + Increment(std::uint64_t id, std::u32string spelling) -> Core::Statement + { + return Core::Statement::Assign( + { id, spelling }, + Core::Expression::InvokePrimitive( + Core::Primitive::Add, + { Variable(id, spelling), Integer(1) }, + Core::Type::int64())); + } + + [[nodiscard]] auto + Accumulate() -> Core::Statement + { + return Core::Statement::Assign( + { kTotal, U"total" }, + Core::Expression::InvokePrimitive( + Core::Primitive::Add, + { Variable(kTotal, U"total"), Variable(kIndex, U"index") }, + Core::Type::int64())); + } + + /// `int total = 0; int index = 0; ; return total;` + [[nodiscard]] auto + LoopModule(Core::Statement loop) -> Core::Module + { + Core::Function function{ + { 1U, U"Evaluate" }, + {}, + Core::Type::int64(), + { Core::Statement::Bind({ { kTotal, U"total" }, + Core::Type::int64(), + true, + Integer(0) }), + Core::Statement::Bind({ { kIndex, U"index" }, + Core::Type::int64(), + true, + Integer(0) }), + std::move(loop), + Core::Statement::Return(Variable(kTotal, U"total")) }, + }; + return { { U"Loops" }, { std::move(function) } }; + } + + [[nodiscard]] auto + PrepareVerified(const Core::Module &module) -> Prepared::Function + { + for (const auto &issue : Core::Verify(module)) + FAIL_CHECK("Core " << issue.code << ": " << issue.message); + REQUIRE(Core::Verify(module).empty()); + auto prepared = Core::CorePrep::Prepare(module); + for (const auto &issue : Prepared::verify(prepared)) + FAIL_CHECK("CorePrep " << issue.code << ": " << issue.message + << " (block " << issue.block << ")"); + REQUIRE(Prepared::verify(prepared).empty()); + REQUIRE(prepared.functions.size() == 1U); + return std::move(prepared.functions.front()); + } + + [[nodiscard]] auto + Find(const Prepared::Function &function, Prepared::BlockId id) + -> const Prepared::Block & + { + const auto found + = std::ranges::find(function.blocks, id, &Prepared::Block::id); + REQUIRE(found != function.blocks.end()); + return *found; + } + + [[nodiscard]] auto + Successors(const Prepared::Block &block) -> std::vector + { + switch (block.terminator.kind) + { + case Prepared::Terminator::Kind::Jump: + return { block.terminator.true_target }; + case Prepared::Terminator::Kind::Branch: + return { block.terminator.true_target, + block.terminator.false_target }; + default: + return {}; + } + } + + /// Blocks reachable from the entry that end in a function return. + [[nodiscard]] auto + ReachesReturn(const Prepared::Function &function, Prepared::BlockId from) + -> bool + { + std::vector pending{ from }; + std::vector seen; + while (!pending.empty()) + { + const auto id = pending.back(); + pending.pop_back(); + if (std::ranges::find(seen, id) != seen.end()) + continue; + seen.push_back(id); + const auto &block = Find(function, id); + if (block.terminator.kind == Prepared::Terminator::Kind::Return) + return true; + for (const auto successor : Successors(block)) + pending.push_back(successor); + } + return false; + } + + [[nodiscard]] auto + IsJumpTo(const Prepared::Block &block, Prepared::BlockId target) -> bool + { + return block.terminator.kind == Prepared::Terminator::Kind::Jump + && block.terminator.true_target == target; + } + + struct ForShape final + { + Prepared::BlockId condition; + Prepared::BlockId body; + Prepared::BlockId update; + Prepared::BlockId exit; + }; + + /// Recover the for-loop block roles from edges alone, then check each. + [[nodiscard]] auto + ForShapeOf(const Prepared::Function &function) -> ForShape + { + const auto &entry = Find(function, function.entry); + REQUIRE(entry.terminator.kind == Prepared::Terminator::Kind::Jump); + const auto conditionId = entry.terminator.true_target; + const auto &condition = Find(function, conditionId); + REQUIRE(condition.terminator.kind + == Prepared::Terminator::Kind::Branch); + const auto bodyId = condition.terminator.true_target; + const auto exitId = condition.terminator.false_target; + // The adapter numbers the update region directly after the body + // entry; the edge checks below confirm that role independently. + return { conditionId, bodyId, bodyId + 1U, exitId }; + } +} // namespace + +TEST_CASE("for-loop update region returns to the condition block", + "[coreprep][loop]") +{ + const auto function = PrepareVerified(LoopModule( + Core::Statement::For(Compare(Core::Primitive::LessThan, kIndex, 1), + { Accumulate() }, + { Increment(kIndex, U"index") }))); + const auto shape = ForShapeOf(function); + + const auto &body = Find(function, shape.body); + const auto &update = Find(function, shape.update); + CHECK(IsJumpTo(body, shape.update)); + // The update region's only successor is the condition. A jump back to + // the update block itself is an infinite loop that skips the condition. + CHECK(IsJumpTo(update, shape.condition)); + CHECK_FALSE(IsJumpTo(update, shape.update)); + CHECK(update.instructions.size() >= 1U); + CHECK(Find(function, shape.exit).terminator.kind + == Prepared::Terminator::Kind::Return); + for (const auto &block : function.blocks) + CHECK(ReachesReturn(function, block.id)); +} + +TEST_CASE("for-loop continue runs the update and break leaves the loop", + "[coreprep][loop]") +{ + const auto function = PrepareVerified(LoopModule(Core::Statement::For( + Compare(Core::Primitive::LessThan, kIndex, 12), + { Core::Statement::If(Compare(Core::Primitive::Equal, kIndex, 2), + { Core::Statement::Continue() }, + {}), + Core::Statement::If(Compare(Core::Primitive::Equal, kIndex, 9), + { Core::Statement::Break() }, + {}), + Accumulate() }, + { Increment(kIndex, U"index") }))); + const auto shape = ForShapeOf(function); + + // Body entry tests `index == 2`; its true arm is the continue. + const auto &body = Find(function, shape.body); + REQUIRE(body.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(IsJumpTo(Find(function, body.terminator.true_target), shape.update)); + + // The join of the first `if` tests `index == 9`; its true arm breaks. + const auto &firstElse = Find(function, body.terminator.false_target); + REQUIRE(firstElse.terminator.kind == Prepared::Terminator::Kind::Jump); + const auto &second = Find(function, firstElse.terminator.true_target); + REQUIRE(second.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(IsJumpTo(Find(function, second.terminator.true_target), shape.exit)); + + const auto &update = Find(function, shape.update); + CHECK(IsJumpTo(update, shape.condition)); + + // Exactly the body tail and the continue arm enter the update region, + // and only the update region re-enters the condition from inside. + std::size_t intoUpdate{}; + std::size_t intoCondition{}; + for (const auto &block : function.blocks) + { + const auto successors = Successors(block); + intoUpdate += static_cast( + std::ranges::count(successors, shape.update)); + intoCondition += static_cast( + std::ranges::count(successors, shape.condition)); + CHECK(ReachesReturn(function, block.id)); + } + CHECK(intoUpdate == 2U); + CHECK(intoCondition == 2U); // function entry and the update region +} + +TEST_CASE("for-loop with an empty update still returns to the condition", + "[coreprep][loop]") +{ + const auto function = PrepareVerified(LoopModule( + Core::Statement::For(Compare(Core::Primitive::LessThan, kIndex, 3), + { Accumulate(), Increment(kIndex, U"index") }, + {}))); + const auto shape = ForShapeOf(function); + const auto &update = Find(function, shape.update); + CHECK(update.instructions.empty()); + CHECK(IsJumpTo(update, shape.condition)); +} + +TEST_CASE("for-loop update containing a branch closes every open tail", + "[coreprep][loop]") +{ + // Structured updates may lower to several blocks; each open tail must + // reach the condition, not only the first update block. + const auto function = PrepareVerified(LoopModule(Core::Statement::For( + Compare(Core::Primitive::LessThan, kIndex, 4), + { Accumulate() }, + { Core::Statement::If( + Compare(Core::Primitive::LessThan, kTotal, 2), + { Increment(kIndex, U"index") }, + { Increment(kIndex, U"index"), Increment(kTotal, U"total") }) }))); + const auto shape = ForShapeOf(function); + const auto &update = Find(function, shape.update); + REQUIRE(update.terminator.kind == Prepared::Terminator::Kind::Branch); + const auto &join = Find( + function, + Find(function, update.terminator.true_target).terminator.true_target); + CHECK(IsJumpTo(join, shape.condition)); + for (const auto &block : function.blocks) + { + CHECK_FALSE(IsJumpTo(block, block.id)); + CHECK(ReachesReturn(function, block.id)); + } +} + +TEST_CASE("for-loop numeric condition keeps its comparison in the header", + "[coreprep][loop]") +{ + // A non-Boolean condition is canonicalized to `value != 0`. That + // comparison belongs to the condition block so every iteration + // re-evaluates it and the branch operand is defined on all paths. + const auto function = PrepareVerified(LoopModule( + Core::Statement::For(Core::Expression::InvokePrimitive( + Core::Primitive::Subtract, + { Integer(3), Variable(kIndex, U"index") }, + Core::Type::int64()), + { Accumulate() }, + { Increment(kIndex, U"index") }))); + const auto shape = ForShapeOf(function); + const auto &condition = Find(function, shape.condition); + REQUIRE(condition.terminator.value.kind == Prepared::Atom::Kind::Variable); + const auto branchSymbol = condition.terminator.value.symbol.id; + CHECK(condition.terminator.value.type == Core::Type::boolean()); + CHECK(std::ranges::any_of( + condition.instructions, + [branchSymbol](const auto &instruction) { + return instruction.destination.id == branchSymbol + && instruction.operation == Prepared::Operation::NotEqual; + })); +} + +TEST_CASE("nested for-loops keep independent continue and break targets", + "[coreprep][loop]") +{ + constexpr std::uint64_t kInner = 4U; + auto inner = Core::Statement::For( + Compare(Core::Primitive::LessThan, kInner, 3), + { Core::Statement::If(Compare(Core::Primitive::Equal, kInner, 1), + { Core::Statement::Continue() }, + {}), + Accumulate() }, + { Increment(kInner, U"inner") }); + const auto function = PrepareVerified(LoopModule(Core::Statement::For( + Compare(Core::Primitive::LessThan, kIndex, 3), + { Core::Statement::Bind( + { { kInner, U"inner" }, Core::Type::int64(), true, Integer(0) }), + std::move(inner) }, + { Increment(kIndex, U"index") }))); + const auto outer = ForShapeOf(function); + + const auto &outerBody = Find(function, outer.body); + REQUIRE(outerBody.terminator.kind == Prepared::Terminator::Kind::Jump); + const auto innerConditionId = outerBody.terminator.true_target; + const auto &innerCondition = Find(function, innerConditionId); + REQUIRE(innerCondition.terminator.kind + == Prepared::Terminator::Kind::Branch); + const auto innerBodyId = innerCondition.terminator.true_target; + const auto innerUpdateId = innerBodyId + 1U; + const auto innerExitId = innerCondition.terminator.false_target; + + CHECK(IsJumpTo(Find(function, innerUpdateId), innerConditionId)); + // The inner continue targets the inner update, never the outer one. + const auto &innerBody = Find(function, innerBodyId); + REQUIRE(innerBody.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(IsJumpTo(Find(function, innerBody.terminator.true_target), + innerUpdateId)); + // Leaving the inner loop falls through to the outer update region. + CHECK(IsJumpTo(Find(function, innerExitId), outer.update)); + CHECK(IsJumpTo(Find(function, outer.update), outer.condition)); + for (const auto &block : function.blocks) + CHECK(ReachesReturn(function, block.id)); +} + +TEST_CASE("while-loop condition owns a header separate from prior statements", + "[coreprep][loop]") +{ + const auto function = PrepareVerified(LoopModule(Core::Statement::While( + Compare(Core::Primitive::LessThan, kIndex, 3), + { Core::Statement::If( + Compare(Core::Primitive::Equal, kIndex, 1), + { Increment(kIndex, U"index"), Core::Statement::Continue() }, + {}), + Accumulate(), + Increment(kIndex, U"index") }))); + + const auto &entry = Find(function, function.entry); + REQUIRE(entry.terminator.kind == Prepared::Terminator::Kind::Jump); + const auto headerId = entry.terminator.true_target; + REQUIRE(headerId != function.entry); + const auto &header = Find(function, headerId); + REQUIRE(header.terminator.kind == Prepared::Terminator::Kind::Branch); + + // The initializers stay in the entry block and are never re-executed: + // nothing in the header may define or assign a source variable. + CHECK(entry.instructions.size() == 2U); + CHECK( + std::ranges::none_of(header.instructions, [](const auto &instruction) { + return instruction.destination.id == kTotal + || instruction.destination.id == kIndex; + })); + + std::size_t backEdges{}; + for (const auto &block : function.blocks) + { + if (block.id != function.entry) + backEdges += static_cast( + std::ranges::count(Successors(block), headerId)); + CHECK(std::ranges::count(Successors(block), function.entry) == 0); + CHECK(ReachesReturn(function, block.id)); + } + CHECK(backEdges == 2U); // the continue arm and the body tail + + const auto &body = Find(function, header.terminator.true_target); + REQUIRE(body.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(IsJumpTo(Find(function, body.terminator.true_target), headerId)); +} + +TEST_CASE("do-while continue targets the trailing condition block", + "[coreprep][loop]") +{ + const auto function = PrepareVerified(LoopModule(Core::Statement::DoWhile( + { Increment(kIndex, U"index"), + Core::Statement::If(Compare(Core::Primitive::Equal, kIndex, 2), + { Core::Statement::Continue() }, + {}), + Core::Statement::If(Compare(Core::Primitive::Equal, kIndex, 5), + { Core::Statement::Break() }, + {}), + Accumulate() }, + Compare(Core::Primitive::LessThan, kIndex, 8)))); + + const auto &entry = Find(function, function.entry); + REQUIRE(entry.terminator.kind == Prepared::Terminator::Kind::Jump); + const auto bodyId = entry.terminator.true_target; + const auto conditionId = bodyId + 1U; + const auto &condition = Find(function, conditionId); + REQUIRE(condition.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(condition.terminator.true_target == bodyId); + const auto exitId = condition.terminator.false_target; + + const auto &body = Find(function, bodyId); + REQUIRE(body.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(IsJumpTo(Find(function, body.terminator.true_target), conditionId)); + const auto &second = Find( + function, + Find(function, body.terminator.false_target).terminator.true_target); + REQUIRE(second.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(IsJumpTo(Find(function, second.terminator.true_target), exitId)); + for (const auto &block : function.blocks) + CHECK(ReachesReturn(function, block.id)); +} diff --git a/Compiler/Core/Tests/ShortCircuitLoweringTests.cpp b/Compiler/Core/Tests/ShortCircuitLoweringTests.cpp new file mode 100644 index 00000000..1e83ec96 --- /dev/null +++ b/Compiler/Core/Tests/ShortCircuitLoweringTests.cpp @@ -0,0 +1,327 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Core/CorePrep/Prepare.hpp" +#include "Visual/XSharp/Core/CorePrep/Verifier.hpp" +#include "Visual/XSharp/Core/Verifier.hpp" + +// `&&` and `||` are control flow in CorePrep: the right operand is reached +// only through the branch that needs it. These tests pin that shape on the +// native Core-to-CorePrep adapter. An eager two-operand instruction computes +// the same Boolean for pure operands but evaluates a guarded division, call +// or recursion that the source never executes. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Prepared = visual_xsharp::core; + + constexpr std::uint64_t kLeft = 2U; + constexpr std::uint64_t kRight = 3U; + constexpr std::uint64_t kThird = 4U; + + [[nodiscard]] auto + Spelling(std::uint64_t id) -> std::u32string + { + return id == kLeft ? U"left" : id == kRight ? U"right" : U"third"; + } + + [[nodiscard]] auto + Integer(std::int64_t value) -> Core::Expression + { + return Core::Expression::Constant(value, Core::Type::int64()); + } + + [[nodiscard]] auto + Variable(std::uint64_t id) -> Core::Expression + { + return Core::Expression::Variable({ id, Spelling(id) }, + Core::Type::int64()); + } + + /// `id < 10`; the distinct symbol identifies which operand a block owns. + [[nodiscard]] auto + Below(std::uint64_t id) -> Core::Expression + { + return Core::Expression::InvokePrimitive(Core::Primitive::LessThan, + { Variable(id), Integer(10) }, + Core::Type::boolean()); + } + + [[nodiscard]] auto + Logical(Core::Primitive operation, + Core::Expression left, + Core::Expression right) -> Core::Expression + { + return Core::Expression::InvokePrimitive( + operation, + { std::move(left), std::move(right) }, + Core::Type::boolean()); + } + + [[nodiscard]] auto + Declare(std::uint64_t id) -> Core::Statement + { + return Core::Statement::Bind( + { { id, Spelling(id) }, Core::Type::int64(), true, Integer(0) }); + } + + [[nodiscard]] auto + Module(Core::Type returnType, std::vector tail) + -> Core::Module + { + std::vector body{ Declare(kLeft), + Declare(kRight), + Declare(kThird) }; + body.insert(body.end(), + std::make_move_iterator(tail.begin()), + std::make_move_iterator(tail.end())); + return { { U"ShortCircuit" }, + { Core::Function{ { 1U, U"Evaluate" }, + {}, + std::move(returnType), + std::move(body) } } }; + } + + [[nodiscard]] auto + PrepareVerified(const Core::Module &module) -> Prepared::Function + { + for (const auto &issue : Core::Verify(module)) + FAIL_CHECK("Core " << issue.code << ": " << issue.message); + REQUIRE(Core::Verify(module).empty()); + auto prepared = Core::CorePrep::Prepare(module); + for (const auto &issue : Prepared::verify(prepared)) + FAIL_CHECK("CorePrep " << issue.code << ": " << issue.message + << " (block " << issue.block << ")"); + REQUIRE(Prepared::verify(prepared).empty()); + REQUIRE(prepared.functions.size() == 1U); + return std::move(prepared.functions.front()); + } + + [[nodiscard]] auto + Find(const Prepared::Function &function, Prepared::BlockId id) + -> const Prepared::Block & + { + const auto found + = std::ranges::find(function.blocks, id, &Prepared::Block::id); + REQUIRE(found != function.blocks.end()); + return *found; + } + + /// The block whose instructions read the given source variable. + [[nodiscard]] auto + BlockReading(const Prepared::Function &function, std::uint64_t symbol) + -> std::optional + { + for (const auto &block : function.blocks) + for (const auto &instruction : block.instructions) + if (instruction.operation == Prepared::Operation::LessThan + && std::ranges::any_of( + instruction.operands, + [symbol](const auto &operand) { + return operand.kind + == Prepared::Atom::Kind::Variable + && operand.symbol.id == symbol; + })) + return block.id; + return std::nullopt; + } + + [[nodiscard]] auto + HasEagerLogicalOperation(const Prepared::Function &function) -> bool + { + return std::ranges::any_of(function.blocks, [](const auto &block) { + return std::ranges::any_of( + block.instructions, + [](const auto &instruction) { + return instruction.operation + == Prepared::Operation::LogicalAnd + || instruction.operation + == Prepared::Operation::LogicalOr; + }); + }); + } + + [[nodiscard]] auto + IsJumpTo(const Prepared::Block &block, Prepared::BlockId target) -> bool + { + return block.terminator.kind == Prepared::Terminator::Kind::Jump + && block.terminator.true_target == target; + } +} // namespace + +TEST_CASE("logical and evaluates its right operand only on the true edge", + "[coreprep][shortcircuit]") +{ + const auto function = PrepareVerified( + Module(Core::Type::boolean(), + { Core::Statement::Return(Logical(Core::Primitive::LogicalAnd, + Below(kLeft), + Below(kRight))) })); + CHECK_FALSE(HasEagerLogicalOperation(function)); + + const auto &entry = Find(function, function.entry); + REQUIRE(entry.terminator.kind == Prepared::Terminator::Kind::Branch); + const auto leftBlock = BlockReading(function, kLeft); + const auto rightBlock = BlockReading(function, kRight); + REQUIRE(leftBlock); + REQUIRE(rightBlock); + CHECK(*leftBlock == function.entry); + CHECK(*rightBlock != function.entry); + // True continues into the right operand; false skips it to the join. + CHECK(entry.terminator.true_target == *rightBlock); + const auto joinId = entry.terminator.false_target; + CHECK(joinId != *rightBlock); + const auto &right = Find(function, *rightBlock); + CHECK(IsJumpTo(right, joinId)); + + // The join returns one slot: initialized false before the branch and + // overwritten only by the right operand's block. + const auto &join = Find(function, joinId); + REQUIRE(join.terminator.kind == Prepared::Terminator::Kind::Return); + REQUIRE(join.terminator.value.kind == Prepared::Atom::Kind::Variable); + const auto result = join.terminator.value.symbol.id; + const auto initializer + = std::ranges::find_if(entry.instructions, + [result](const auto &instruction) { + return instruction.destination.id == result; + }); + REQUIRE(initializer != entry.instructions.end()); + CHECK(initializer->kind == Prepared::Instruction::Kind::Bind); + CHECK(initializer->mutable_binding); + REQUIRE(initializer->operands.size() == 1U); + CHECK(initializer->operands.front().literal == Prepared::Literal{ false }); + REQUIRE_FALSE(right.instructions.empty()); + CHECK(right.instructions.back().kind + == Prepared::Instruction::Kind::Assign); + CHECK(right.instructions.back().destination.id == result); +} + +TEST_CASE("logical or evaluates its right operand only on the false edge", + "[coreprep][shortcircuit]") +{ + const auto function = PrepareVerified( + Module(Core::Type::boolean(), + { Core::Statement::Return(Logical(Core::Primitive::LogicalOr, + Below(kLeft), + Below(kRight))) })); + CHECK_FALSE(HasEagerLogicalOperation(function)); + + const auto &entry = Find(function, function.entry); + REQUIRE(entry.terminator.kind == Prepared::Terminator::Kind::Branch); + const auto rightBlock = BlockReading(function, kRight); + REQUIRE(rightBlock); + CHECK(entry.terminator.false_target == *rightBlock); + const auto joinId = entry.terminator.true_target; + CHECK(IsJumpTo(Find(function, *rightBlock), joinId)); + const auto &join = Find(function, joinId); + REQUIRE(join.terminator.kind == Prepared::Terminator::Kind::Return); + const auto result = join.terminator.value.symbol.id; + const auto initializer + = std::ranges::find_if(entry.instructions, + [result](const auto &instruction) { + return instruction.destination.id == result; + }); + REQUIRE(initializer != entry.instructions.end()); + REQUIRE(initializer->operands.size() == 1U); + CHECK(initializer->operands.front().literal == Prepared::Literal{ true }); +} + +TEST_CASE("nested short-circuit operands stay behind their own guards", + "[coreprep][shortcircuit]") +{ + // (left && right) || third: `right` needs left, `third` needs the + // conjunction to be false, and neither is evaluated in the entry block. + const auto function = PrepareVerified(Module( + Core::Type::boolean(), + { Core::Statement::Return(Logical( + Core::Primitive::LogicalOr, + Logical(Core::Primitive::LogicalAnd, Below(kLeft), Below(kRight)), + Below(kThird))) })); + CHECK_FALSE(HasEagerLogicalOperation(function)); + const auto leftBlock = BlockReading(function, kLeft); + const auto rightBlock = BlockReading(function, kRight); + const auto thirdBlock = BlockReading(function, kThird); + REQUIRE(leftBlock); + REQUIRE(rightBlock); + REQUIRE(thirdBlock); + CHECK(*leftBlock == function.entry); + CHECK(*rightBlock != function.entry); + CHECK(*thirdBlock != function.entry); + CHECK(*thirdBlock != *rightBlock); + + const auto &entry = Find(function, function.entry); + REQUIRE(entry.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(entry.terminator.true_target == *rightBlock); + // The conjunction's join decides whether `third` runs at all. + const auto &innerJoin = Find(function, entry.terminator.false_target); + REQUIRE(innerJoin.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(innerJoin.terminator.false_target == *thirdBlock); + CHECK(IsJumpTo(Find(function, *thirdBlock), + innerJoin.terminator.true_target)); +} + +TEST_CASE("short-circuit loop condition is re-evaluated from the loop header", + "[coreprep][shortcircuit][loop]") +{ + const auto function = PrepareVerified(Module( + Core::Type::int64(), + { Core::Statement::While( + Logical(Core::Primitive::LogicalAnd, Below(kLeft), Below(kRight)), + { Core::Statement::Assign({ kLeft, Spelling(kLeft) }, + Core::Expression::InvokePrimitive( + Core::Primitive::Add, + { Variable(kLeft), Integer(1) }, + Core::Type::int64())) }), + Core::Statement::Return(Variable(kLeft)) })); + CHECK_FALSE(HasEagerLogicalOperation(function)); + + const auto &entry = Find(function, function.entry); + REQUIRE(entry.terminator.kind == Prepared::Terminator::Kind::Jump); + const auto headerId = entry.terminator.true_target; + const auto leftBlock = BlockReading(function, kLeft); + const auto rightBlock = BlockReading(function, kRight); + REQUIRE(leftBlock); + REQUIRE(rightBlock); + CHECK(*leftBlock == headerId); + CHECK(*rightBlock != headerId); + + // The header only decides whether the right operand runs; the loop's + // own body/exit branch lives in the short-circuit join. + const auto &header = Find(function, headerId); + REQUIRE(header.terminator.kind == Prepared::Terminator::Kind::Branch); + CHECK(header.terminator.true_target == *rightBlock); + const auto &join = Find(function, header.terminator.false_target); + REQUIRE(join.terminator.kind == Prepared::Terminator::Kind::Branch); + const auto &body = Find(function, join.terminator.true_target); + // The back-edge re-enters the header so `left` is tested again. + CHECK(IsJumpTo(body, headerId)); + CHECK(Find(function, join.terminator.false_target).terminator.kind + == Prepared::Terminator::Kind::Return); +} + +TEST_CASE("logical not remains an ordinary single-block operation", + "[coreprep][shortcircuit]") +{ + const auto function = PrepareVerified( + Module(Core::Type::boolean(), + { Core::Statement::Return(Core::Expression::InvokePrimitive( + Core::Primitive::LogicalNot, + { Below(kLeft) }, + Core::Type::boolean())) })); + REQUIRE(function.blocks.size() == 1U); + CHECK(std::ranges::any_of(function.blocks.front().instructions, + [](const auto &instruction) { + return instruction.operation + == Prepared::Operation::LogicalNot; + })); +} diff --git a/Compiler/Core/Tests/SymbolAllocationTests.cpp b/Compiler/Core/Tests/SymbolAllocationTests.cpp new file mode 100644 index 00000000..a7a18f6b --- /dev/null +++ b/Compiler/Core/Tests/SymbolAllocationTests.cpp @@ -0,0 +1,197 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Core/CorePrep/Prepare.hpp" +#include "Visual/XSharp/Core/CorePrep/Verifier.hpp" +#include "Visual/XSharp/Core/Verifier.hpp" +#include "Visual/XSharp/Core/Wire.hpp" +#include "Visual/XSharp/Pipeline.hpp" + +// Symbol identities are unique across a whole module, and CorePrep +// verification checks that one identity never carries two spellings. The +// adapter's generated temporaries and lifted closure names must therefore be +// allocated above every identity of the module from one shared counter, not +// above the identities of the function being lowered. These cases were found +// by the source-to-LLVM fuzz campaign as a verified Core module that the +// native pipeline then rejected. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Prepared = visual_xsharp::core; + + [[nodiscard]] auto + Integer(std::int64_t value) -> Core::Expression + { + return Core::Expression::Constant(value, Core::Type::int64()); + } + + [[nodiscard]] auto + Variable(std::uint64_t id, std::u32string spelling) -> Core::Expression + { + return Core::Expression::Variable({ id, std::move(spelling) }, + Core::Type::int64()); + } + + /// `int result = 8; if (result > 4) { result = result - 2; } return + /// result;` The comparison and the subtraction each need a generated + /// temporary. + [[nodiscard]] auto + Branching(std::uint64_t function, std::uint64_t local, std::u32string name) + -> Core::Function + { + return { + { function, std::move(name) }, + {}, + Core::Type::int64(), + { Core::Statement::Bind({ { local, U"result" }, + Core::Type::int64(), + true, + Integer(8) }), + Core::Statement::If( + Core::Expression::InvokePrimitive( + Core::Primitive::GreaterThan, + { Variable(local, U"result"), Integer(4) }, + Core::Type::boolean()), + { Core::Statement::Assign( + { local, U"result" }, + Core::Expression::InvokePrimitive( + Core::Primitive::Subtract, + { Variable(local, U"result"), Integer(2) }, + Core::Type::int64())) }, + {}), + Core::Statement::Return(Variable(local, U"result")) }, + }; + } + + /// Every identity in the prepared module with each spelling it carries. + void + Record(std::map> &seen, + const Prepared::SymbolName &symbol) + { + if (symbol.id == 0U) + return; + auto &spellings = seen[symbol.id]; + if (std::ranges::find(spellings, symbol.spelling) == spellings.end()) + spellings.push_back(symbol.spelling); + } + + void + RequireUniqueIdentities(const Prepared::CorePrepModule &module) + { + std::map> seen; + for (const auto &function : module.functions) + { + Record(seen, function.symbol); + for (const auto ¶meter : function.parameters) + Record(seen, parameter.symbol); + for (const auto &block : function.blocks) + for (const auto &instruction : block.instructions) + { + Record(seen, instruction.destination); + Record(seen, instruction.closure_function); + for (const auto &operand : instruction.operands) + if (operand.kind == Prepared::Atom::Kind::Variable) + Record(seen, operand.symbol); + } + } + for (const auto &[id, spellings] : seen) + { + CAPTURE(id); + CHECK(spellings.size() == 1U); + } + } + + void + RequireVerifiedPipeline(const Core::Module &module) + { + for (const auto &issue : Core::Verify(module)) + FAIL_CHECK("Core " << issue.code << ": " << issue.message); + REQUIRE(Core::Verify(module).empty()); + const auto prepared = Core::CorePrep::Prepare(module); + for (const auto &issue : Prepared::verify(prepared)) + FAIL_CHECK("CorePrep " << issue.code << ": " << issue.message); + CHECK(Prepared::verify(prepared).empty()); + RequireUniqueIdentities(prepared); + const auto encoded = Core::Wire::Encode(module); + REQUIRE(encoded); + const auto pipeline + = Visual::XSharp::Pipeline::ConsumeCore(encoded.bytes); + CHECK(pipeline); + CHECK(pipeline.llvm); + } +} // namespace + +TEST_CASE("temporaries of one function do not reuse another function's " + "identities", + "[coreprep][symbols]") +{ + // Identities 1 and 2 belong to the first function. Its first temporary + // would be 3 if allocation only looked at that function, which is the + // second function's own name. + RequireVerifiedPipeline( + { { U"Symbols" }, + { Branching(1U, 2U, U"First"), Branching(3U, 4U, U"Second") } }); +} + +TEST_CASE("temporaries stay unique when a later function has lower " + "identities", + "[coreprep][symbols]") +{ + RequireVerifiedPipeline( + { { U"Symbols" }, + { Branching(10U, 11U, U"First"), Branching(1U, 2U, U"Second") } }); +} + +TEST_CASE("lifted closure names and temporaries share one identity counter", + "[coreprep][symbols]") +{ + // The closure is lifted to a generated function symbol while the + // enclosing body also needs temporaries for its arithmetic. + const auto closureType + = Core::Type::function({ Core::Type::int64() }, Core::Type::int64()); + Core::Function function{ + { 1U, U"Evaluate" }, + {}, + Core::Type::int64(), + { Core::Statement::Bind( + { { 2U, U"base" }, Core::Type::int64(), false, Integer(5) }), + Core::Statement::Bind( + { { 3U, U"scale" }, + closureType, + false, + Core::Expression::Closure( + {}, + { { { 4U, U"value" }, Core::Type::int64() } }, + Core::Type::int64(), + { Core::Statement::Return(Core::Expression::InvokePrimitive( + Core::Primitive::Multiply, + { Core::Expression::InvokePrimitive( + Core::Primitive::Add, + { Variable(4U, U"value"), Integer(1) }, + Core::Type::int64()), + Integer(2) }, + Core::Type::int64())) }, + closureType) }), + Core::Statement::Return(Core::Expression::InvokePrimitive( + Core::Primitive::Add, + { Core::Expression::InvokePrimitive( + Core::Primitive::Multiply, + { Variable(2U, U"base"), Integer(3) }, + Core::Type::int64()), + Core::Expression::Apply( + Core::Expression::Variable({ 3U, U"scale" }, closureType), + { Integer(4) }, + Core::Type::int64()) }, + Core::Type::int64())) }, + }; + RequireVerifiedPipeline({ { U"Symbols" }, { std::move(function) } }); +} diff --git a/Compiler/Fuzzing/BUILD.bazel b/Compiler/Fuzzing/BUILD.bazel index 0fbbce17..0d366e82 100644 --- a/Compiler/Fuzzing/BUILD.bazel +++ b/Compiler/Fuzzing/BUILD.bazel @@ -70,3 +70,10 @@ cc_binary( defines = ["VXS_SOURCE_FUZZ_STAGE=2"], deps = [":source_fuzz_harness"], ) + +cc_binary( + name = "differential_fuzzer", + srcs = ["SourceLibFuzzer.cpp"], + defines = ["VXS_SOURCE_FUZZ_STAGE=3"], + deps = [":source_fuzz_harness"], +) diff --git a/Compiler/Fuzzing/Corpus/cli/options.seed b/Compiler/Fuzzing/Corpus/cli/options.seed new file mode 100644 index 00000000..c3081393 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/cli/options.seed @@ -0,0 +1,3 @@ +build +-Help +-- diff --git a/Compiler/Fuzzing/Corpus/differential/arithmetic.seed b/Compiler/Fuzzing/Corpus/differential/arithmetic.seed new file mode 100644 index 00000000..55a38a55 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/differential/arithmetic.seed @@ -0,0 +1 @@ +arithmetic-oracle-seed diff --git a/Compiler/Fuzzing/Corpus/ownership/race.seed b/Compiler/Fuzzing/Corpus/ownership/race.seed new file mode 100644 index 00000000..30f4b487 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/ownership/race.seed @@ -0,0 +1 @@ +strong-weak-unowned-race diff --git a/Compiler/Fuzzing/Corpus/project/registry.seed b/Compiler/Fuzzing/Corpus/project/registry.seed new file mode 100644 index 00000000..72bcbb43 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/project/registry.seed @@ -0,0 +1 @@ +visual-xsharp-sources-v6 diff --git a/Compiler/Fuzzing/Corpus/repl/history.seed b/Compiler/Fuzzing/Corpus/repl/history.seed new file mode 100644 index 00000000..07ea8213 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/repl/history.seed @@ -0,0 +1 @@ +repl-session-reset-type-error-value diff --git a/Compiler/Fuzzing/Corpus/source/cross-function-temporaries.seed b/Compiler/Fuzzing/Corpus/source/cross-function-temporaries.seed new file mode 100644 index 00000000..f727f2c0 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/source/cross-function-temporaries.seed @@ -0,0 +1,2 @@ +namespace Fuzz; class Progrfm { public static long E6aluate() { long result = 8; if (result +> 4) { result = result - 2; } return result; } } class Program { public static long Evannnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnnluate() { return 0 + 2 * 3; } } diff --git a/Compiler/Fuzzing/Corpus/source/reserved-llvm-namespace.seed b/Compiler/Fuzzing/Corpus/source/reserved-llvm-namespace.seed new file mode 100644 index 00000000..380b2fc3 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/source/reserved-llvm-namespace.seed @@ -0,0 +1 @@ +namespace llvm.FLLLLLLLQQQQ1QQQQQQQQQQQQQQQQQQQQQQQQQQQQQQQQQQQQQQLLLLuzz; class Program { public static long Evaluate() { long result = 6; if (result >3) { result = result - 2; } return result; } } diff --git a/Compiler/Fuzzing/SourceFuzz.cpp b/Compiler/Fuzzing/SourceFuzz.cpp index 55b7aab2..009250b8 100644 --- a/Compiler/Fuzzing/SourceFuzz.cpp +++ b/Compiler/Fuzzing/SourceFuzz.cpp @@ -2,8 +2,10 @@ // SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 #include +#include #include #include +#include #include #include #include @@ -78,12 +80,89 @@ namespace Visual::XSharp::Fuzzing const auto expression = GenerateExpression(bytes, cursor, kMaximumGeneratedDepth); expected = expression.value; + std::string body = "return " + expression.source + ";"; + std::string members; + const auto mode = NextByte(bytes, cursor) % 6U; + const auto limit + = static_cast(NextByte(bytes, cursor) % 12U); + if (mode == 1U) + { + // Independent host execution models for/continue/break rather + // than comparing two copies of the compiler's CFG algorithm. + expected = 0; + for (std::int64_t index = 0; index < limit; ++index) + { + if (index == 2) + continue; + if (index == 9) + break; + expected += index; + } + body = "int total = 0; for (int index = 0; index < " + + std::to_string(limit) + + "; index++) { if (index == 2) { continue; } " + "if (index == 9) { break; } total = total + index; } " + "return total;"; + } + else if (mode == 2U) + { + expected = limit == 0 ? 1 : limit; + body = "int total = 0; do { total++; } while (total < " + + std::to_string(limit) + "); return total;"; + } + else if (mode == 3U) + { + expected = limit < 6 ? expression.value : -expression.value; + body = "if (" + std::to_string(limit) + " < 6) { return " + + expression.source + "; } else { return -(" + + expression.source + "); }"; + } + else if (mode == 4U) + { + // Statements before a pre-test loop run exactly once. The + // initializers and the loop share one source block, so a + // back-edge that re-enters that block resets the counter. + expected = 0; + std::int64_t index = 0; + while (index < limit) + { + if (index == 1) + { + ++index; + continue; + } + if (index == 7) + break; + expected += index; + ++index; + } + body = "int total = 0; int index = 0; while (index < " + + std::to_string(limit) + + ") { if (index == 1) { index++; continue; } " + "if (index == 7) { break; } total = total + index; " + "index++; } return total;"; + } + else if (mode == 5U) + { + // The recursion ends only because `||` skips its right + // operand at zero, and the comparison after `&&` is reached + // only when the call returned. Evaluating either right + // operand eagerly recurses without bound. + expected = limit < 6 ? limit : -1; + members + = " public static bool Down(_ int n) { return n == 0 " + "|| Down(n - 1); }\n"; + body = "if (Down(" + std::to_string(limit) + ") && " + + std::to_string(limit) + " < 6) { return " + + std::to_string(limit) + "; } return 0 - 1;"; + } return "namespace Fuzz;\n" "class Program {\n" - " public static int Evaluate() {\n" - " return " - + expression.source - + ";\n" + + members + + " public static int Evaluate() {\n" + " " + + body + + "\n" " }\n" "}\n"; } @@ -174,23 +253,18 @@ namespace Visual::XSharp::Fuzzing } [[nodiscard]] auto - CompileVariant(std::span source, + CompileVariant(std::span coreBytes, bool optimizeXpp, bool optimizeXmm) -> Llvm::Artifact { - const auto compiled = CompileSource(source); - if (!compiled.succeeded() - || compiled.kind != Frontend::OutputKind::CoreWire) - llvm::report_fatal_error( - llvm::Twine("generated arithmetic source was rejected by " - "the frontend")); - Visual::XSharp::Pipeline::Options options; options.optimize_xpp = optimizeXpp; options.optimize_xmm = optimizeXmm; + options.llvm.optimization = optimizeXmm + ? Llvm::OptimizationLevel::Default + : Llvm::OptimizationLevel::Debug; const auto pipeline - = Visual::XSharp::Pipeline::ConsumeCore(compiled.bytes, - options); + = Visual::XSharp::Pipeline::ConsumeCore(coreBytes, options); if (!pipeline || !pipeline.llvm) llvm::report_fatal_error(llvm::Twine( "differential source failed a verified compiler pipeline: " @@ -288,8 +362,21 @@ namespace Visual::XSharp::Fuzzing const auto bytes = std::span( reinterpret_cast(source.data()), source.size()); - const auto unoptimized = CompileVariant(bytes, false, false); - const auto optimized = CompileVariant(bytes, true, true); + const auto compiled = CompileSource(bytes); + if (!compiled.succeeded() + || compiled.kind != Frontend::OutputKind::CoreWire) + llvm::report_fatal_error( + llvm::Twine("generated arithmetic source was rejected by " + "the frontend")); + // The comparison varies native optimizers, so both paths start from + // the same frontend result. Recompiling identical source adds no + // independent evidence and repeats work in the expensive oracle. + const auto unoptimized = CompileVariant(compiled.bytes, false, false); + const auto optimized = CompileVariant(compiled.bytes, true, true); + if (std::getenv("VXS_FUZZ_TRACE") != nullptr) + llvm::errs() << source << "\nReference LLVM:\n" + << unoptimized.llvm_ir << "\nOptimized LLVM:\n" + << optimized.llvm_ir; constexpr std::string_view kReferenceModule = "vxs-fuzz-reference"; constexpr std::string_view kOptimizedModule = "vxs-fuzz-optimized"; const auto referenceValue = Invoke(unoptimized, kReferenceModule); diff --git a/Compiler/Fuzzing/SourceFuzzSmoke.cpp b/Compiler/Fuzzing/SourceFuzzSmoke.cpp index 72090b04..52ae85bb 100644 --- a/Compiler/Fuzzing/SourceFuzzSmoke.cpp +++ b/Compiler/Fuzzing/SourceFuzzSmoke.cpp @@ -3,6 +3,7 @@ #include #include +#include #include #include "SourceFuzz.hpp" @@ -23,6 +24,37 @@ main() const std::span emptySource; Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(emptySource); Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(source); + llvm::errs() << "Differential smoke: mixed seed\n"; Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(expressionSeed); + llvm::errs() << "Differential smoke: empty seed\n"; + Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(emptySource); + // Constant seeds force leaves, full-depth addition, subtraction and + // multiplication. Exercise the independent oracle before a mutation + // campaign so a missing generated-code route cannot appear as success. + constexpr std::array selectors{ 252U, 253U, 254U, 255U }; + for (const auto selector : selectors) + { + const std::array seed{ selector }; + llvm::errs() << "Differential smoke: selector " + << static_cast(selector) << '\n'; + Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(seed); + } + // A leaf selector followed by an explicit mode and limit byte reaches + // every generated control-flow shape at every trip count, including the + // zero-trip, continue and break paths, instead of only the shapes the + // four cycling selectors above happen to select. + constexpr std::uint8_t kModes = 6U; + constexpr std::uint8_t kLimits = 12U; + for (std::uint8_t mode = 0U; mode < kModes; ++mode) + { + for (std::uint8_t limit = 0U; limit < kLimits; ++limit) + { + const std::array seed{ 0U, mode, limit }; + llvm::errs() << "Differential smoke: mode " + << static_cast(mode) << " limit " + << static_cast(limit) << '\n'; + Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(seed); + } + } return 0; } diff --git a/Compiler/Fuzzing/SourceLibFuzzer.cpp b/Compiler/Fuzzing/SourceLibFuzzer.cpp index bc326475..3f933312 100644 --- a/Compiler/Fuzzing/SourceLibFuzzer.cpp +++ b/Compiler/Fuzzing/SourceLibFuzzer.cpp @@ -20,7 +20,10 @@ LLVMFuzzerTestOneInput(const std::uint8_t *data, std::size_t size) #elif VXS_SOURCE_FUZZ_STAGE == 1 Visual::XSharp::Fuzzing::ExerciseParser(input); #elif VXS_SOURCE_FUZZ_STAGE == 2 + // Arbitrary source and generated arithmetic have independent corpora and + // time budgets. An invalid source mutation should not pay for two JITs. Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(input); +#elif VXS_SOURCE_FUZZ_STAGE == 3 Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(input); #else # error Unsupported VXS_SOURCE_FUZZ_STAGE diff --git a/Compiler/Haskell/Driver/Fuzzing/Feedback.hs b/Compiler/Haskell/Driver/Fuzzing/Feedback.hs new file mode 100644 index 00000000..be3edc02 --- /dev/null +++ b/Compiler/Haskell/Driver/Fuzzing/Feedback.hs @@ -0,0 +1,39 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Exact HPC tick identities for frontend mutation feedback. Module hashes +keep different builds' counters distinct; each execution resets counters before +running so coverage is attributed to that input rather than accumulated history. +-} +module Feedback (Coverage, coverageFor, availableTicks, requiredModulePresent) where + +import Data.List (isInfixOf) +import Data.Set qualified as Set +import Trace.Hpc.Tix (Tix (..), TixModule (..)) + +type Coverage = Set.Set (String, String, Int) + +ownedModule :: String -> Bool +ownedModule = isInfixOf "Visual.XSharp." + +coverageFor :: Tix -> Coverage +coverageFor (Tix modules) = + Set.fromList + [ (name, show identity, index) + | TixModule name identity _ ticks <- modules + , ownedModule name + , (index, count) <- zip [0 ..] ticks + , count > 0 + ] + +availableTicks :: Tix -> Int +availableTicks (Tix modules) = sum [size | TixModule name _ size _ <- modules, ownedModule name] + +requiredModulePresent :: String -> Tix -> Bool +requiredModulePresent stage (Tix modules) = + any (\(TixModule name _ size _) -> size > 0 && isInfixOf expected name) modules + where + expected = case stage of + "lexer" -> "Visual.XSharp.Lexer" + "parser" -> "Visual.XSharp.Parser" + _ -> "Visual.XSharp.Compiler" diff --git a/Compiler/Haskell/Driver/Fuzzing/Main.hs b/Compiler/Haskell/Driver/Fuzzing/Main.hs new file mode 100644 index 00000000..04179fde --- /dev/null +++ b/Compiler/Haskell/Driver/Fuzzing/Main.hs @@ -0,0 +1,182 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +{-# LANGUAGE BangPatterns #-} + +module Main (main) where + +import Control.Exception (SomeException, displayException, evaluate, try) +import Control.Monad (forM, unless, when) +import Data.ByteString qualified as BS +import Data.List (sort) +import Data.Sequence qualified as Seq +import Data.Set qualified as Set +import Data.Text qualified as Text +import Data.Text.Encoding qualified as Text +import Data.Word (Word64) +import Feedback +import GHC.Clock (getMonotonicTimeNSec) +import Mutation +import System.Directory + ( createDirectoryIfMissing + , doesFileExist + , getFileSize + , listDirectory + , pathIsSymbolicLink + , renameFile + ) +import System.Environment (getArgs) +import System.Exit (die) +import System.FilePath (()) +import System.IO (hClose, openBinaryTempFile) +import System.IO.Error (catchIOError, isDoesNotExistError) +import System.Timeout (timeout) +import Text.Read (readMaybe) +import Trace.Hpc.Reflect (clearTix, examineTix) +import Visual.XSharp.Compiler +import Visual.XSharp.Frontend (analyzeSyntax) +import Visual.XSharp.Lexer + +main :: IO () +main = do + arguments <- getArgs + case arguments of + [stage, "--replay", path] -> do + validateStage stage + input <- BS.readFile path + when (BS.length input > maximumInput) (die "replay input exceeds the campaign limit") + execute stage input >>= either die (const (putStrLn "HPC replay completed")) + [stage, secondsText, corpus, artifacts, seedText] + | Just seconds <- readMaybe secondsText + , seconds >= (1 :: Int) && seconds <= 3600 + , Just seed <- readMaybe seedText -> do + validateStage stage + campaign stage seconds corpus artifacts seed + _ -> die "usage: frontend-fuzz STAGE SECONDS CORPUS ARTIFACTS SEED | STAGE --replay FILE" + +validateStage :: String -> IO () +validateStage stage = unless (stage `elem` ["lexer", "parser", "source"]) (die "unknown HPC fuzz stage") + +execute :: String -> BS.ByteString -> IO (Either String Coverage) +execute stage input = do + -- Timeout surrounds exception capture so ordinary lazy failures cannot + -- escape, and a timed-out input cannot be mistaken for a normal diagnostic. + clearTix + attempted <- try (timeout 5000000 action) :: IO (Either SomeException (Maybe Int)) + case attempted of + Left issue -> pure (Left (displayException issue)) + Right Nothing -> pure (Left "frontend input exceeded five seconds") + Right (Just _) -> Right . coverageFor <$> examineTix + where + action = case Text.decodeUtf8' input of + Left _ -> pure 0 + Right text -> do + let source = Text.unpack text + compilerInput = CompilerInput "Fuzz.vxs" source + evaluate $ length $ case stage of + "lexer" -> show (runLexer defaultLexer (LexerInput "Fuzz.vxs" source)) + "parser" -> show (analyzeSyntax compilerInput) + _ -> show (compileToCorePrep compilerInput) + +loadSeeds :: FilePath -> IO [BS.ByteString] +loadSeeds corpus = do + names <- sort <$> listDirectory corpus + inputs <- forM (take 4096 names) $ \name -> do + let path = corpus name + regular <- doesFileExist path + symbolic <- pathIsSymbolicLink path + if regular && not symbolic + then do + size <- getFileSize path + if size <= fromIntegral maximumInput then Just <$> BS.readFile path else pure Nothing + else pure Nothing + pure (BS.empty : [bytes | Just bytes <- inputs]) + +saveInput :: FilePath -> BS.ByteString -> IO () +saveInput corpus input = do + let path = corpus ("hpc-" ++ fingerprint input ++ ".seed") + symbolic <- + pathIsSymbolicLink path `catchIOError` \issue -> + if isDoesNotExistError issue then pure False else ioError issue + when symbolic (die "cached HPC corpus entry is a symbolic link") + exists <- doesFileExist path + if exists + then do + previous <- BS.readFile path + unless (previous == input) (die "HPC corpus fingerprint collision; refusing to replace the existing input") + else do + -- Rename a newly created regular file instead of opening the final + -- cached name for writing. A dangling/replaced symlink must never + -- redirect writes outside the single-writer campaign corpus. + (temporary, handle) <- openBinaryTempFile corpus ".pending-seed" + BS.hPut handle input + hClose handle + renameFile temporary path + +campaign :: String -> Int -> FilePath -> FilePath -> Word64 -> IO () +campaign stage seconds corpus artifacts initialState = do + createDirectoryIfMissing True corpus + createDirectoryIfMissing True artifacts + seeds <- loadSeeds corpus + -- Prime registration with an actually valid source before checking HPC. + -- A coverage-disabled build must fail rather than run unguided mutations. + let validSource = + BS.pack (map (fromIntegral . fromEnum) "namespace Fuzz; class Program { public static int Evaluate() { return 1; } }") + execute stage validSource >>= either die (const (pure ())) + initialTix <- examineTix + unless (requiredModulePresent stage initialTix) (die "frontend-fuzz requires HPC-instrumented production modules") + let tickCount = availableTicks initialTix + started <- getMonotonicTimeNSec + let deadline = started + fromIntegral seconds * 1000000000 + report executed coverage additions = + unlines + [ "stage=" ++ stage + , "executed_units=" ++ show executed + , "covered_ticks=" ++ show (Set.size coverage) + , "available_ticks=" ++ show tickCount + , "new_units_added=" ++ show additions + , "seed=" ++ show initialState + ] + finish executed coverage additions = do + unless (executed > 0 && not (Set.null coverage)) (die "HPC campaign did not execute instrumented production code") + let summary = report executed coverage additions + writeFile (artifacts "campaign.txt") summary + putStrLn ("HPC_FUZZ_RESULT\n" ++ summary) + failInput input problem = do + BS.writeFile (artifacts "failure.seed") input + writeFile (artifacts "failure.txt") (problem ++ "\nReplay: frontend-fuzz " ++ stage ++ " --replay failure.seed\n") + die ("HPC fuzz failure: " ++ problem) + addInput input pool coverage additions = do + checked <- execute stage input + case checked of + Left problem -> failInput input problem + Right observed -> do + let novel = not (observed `Set.isSubsetOf` coverage) + when novel (saveInput corpus input) + let nextPool = if novel && Seq.length pool < 4096 then pool Seq.|> input else pool + pure (nextPool, Set.union observed coverage, additions + if novel then 1 else 0) + warm pool coverage additions [] = pure (pool, coverage, additions) + warm pool coverage additions (input : remaining) = do + (nextPool, nextCoverage, nextAdditions) <- addInput input pool coverage additions + warm nextPool nextCoverage nextAdditions remaining + loop !state !pool !coverage !executed !additions = do + now <- getMonotonicTimeNSec + if now >= deadline + then finish executed coverage additions + else do + let next = nextState state + base = Seq.index pool (fromIntegral (state `mod` fromIntegral (Seq.length pool))) + partner = Seq.index pool (fromIntegral (next `mod` fromIntegral (Seq.length pool))) + input = mutate next base partner + (nextPool, nextCoverage, nextAdditions) <- addInput input pool coverage additions + -- Progress checkpoints retain mutation state for resource + -- termination; ordinary failures always save exact bytes. + when + (executed `mod` 256 == 0) + ( writeFile + (artifacts "progress.txt") + (report executed nextCoverage nextAdditions ++ "mutation_state=" ++ show next ++ "\n") + ) + loop next nextPool nextCoverage (executed + 1) nextAdditions + let warmSeeds = validSource : take 4096 seeds + (pool, coverage, additions) <- warm (Seq.singleton BS.empty) Set.empty (0 :: Int) warmSeeds + loop initialState pool coverage (length warmSeeds) additions diff --git a/Compiler/Haskell/Driver/Fuzzing/Mutation.hs b/Compiler/Haskell/Driver/Fuzzing/Mutation.hs new file mode 100644 index 00000000..0c2d2e77 --- /dev/null +++ b/Compiler/Haskell/Driver/Fuzzing/Mutation.hs @@ -0,0 +1,76 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Bounded deterministic byte mutations. A campaign records its initial state +and saves exact failing bytes, so both mutation sequences and individual failures +are reproducible. Word64 wrapping is intentional for this generator. +-} +module Mutation (mutate, nextState, fingerprint, maximumInput) where + +import Data.Bits (xor) +import Data.ByteString qualified as BS +import Data.ByteString.Char8 qualified as Char8 +import Data.Word (Word64) +import Numeric (showHex) + +maximumInput :: Int +maximumInput = 8192 + +nextState :: Word64 -> Word64 +nextState state = state * 6364136223846793005 + 1442695040888963407 + +fingerprint :: BS.ByteString -> String +fingerprint bytes = + showHex + (BS.foldl' (\hash byte -> (hash `xor` fromIntegral byte) * 1099511628211) (14695981039346656037 :: Word64) bytes) + "" + +dictionary :: [BS.ByteString] +dictionary = + map + Char8.pack + [ "namespace Fuzz;" + , "class Program {" + , "public static int Evaluate() {" + , "return " + , "int value = " + , "while (" + , "if (" + , "for (int i = 0; i < 3; i++) {" + , "break;" + , "continue;" + , "not " + , "\\=" + , "&&" + , "||" + , "**" + , "-- comment\n" + , "0" + , "1" + , "255" + , "65536" + , ";" + , "}" + , "(" + , ")" + , "\"" + , "\0" + , "\r\n" + ] + +mutate :: Word64 -> BS.ByteString -> BS.ByteString -> BS.ByteString +mutate state input partner = BS.take maximumInput result + where + size = BS.length input + position = fromIntegral (nextState state `mod` fromIntegral (size + 1)) + prefix = BS.take position input + suffix = BS.drop position input + byte = fromIntegral (nextState (nextState state) `mod` 256) + token = dictionary !! fromIntegral (nextState state `mod` fromIntegral (length dictionary)) + result = case state `mod` 6 of + 0 -> prefix <> BS.singleton byte <> BS.drop 1 suffix + 1 -> prefix <> token <> suffix + 2 -> prefix <> BS.drop (1 + fromIntegral (state `mod` 16)) suffix + 3 -> prefix <> BS.take 128 partner <> suffix + 4 -> prefix <> BS.take 64 suffix <> suffix + _ -> prefix <> BS.singleton byte <> suffix diff --git a/Compiler/Haskell/Driver/Fuzzing/Tests.hs b/Compiler/Haskell/Driver/Fuzzing/Tests.hs new file mode 100644 index 00000000..beb1b603 --- /dev/null +++ b/Compiler/Haskell/Driver/Fuzzing/Tests.hs @@ -0,0 +1,39 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +module Main (main) where + +import Control.Monad (unless) +import Data.ByteString qualified as BS +import Data.Set qualified as Set +import Feedback +import Mutation +import System.Exit (die) +import Trace.Hpc.Tix (Tix (..), TixModule (..)) +import Trace.Hpc.Util (toHash) + +main :: IO () +main = do + let ticks = Tix [TixModule "Visual.XSharp.Parser" (toHash (1 :: Int)) 3 [1, 0, 2], TixModule "Main" (toHash (2 :: Int)) 1 [1]] + changed = Tix [TixModule "Visual.XSharp.Parser" (toHash (3 :: Int)) 3 [1, 0, 2]] + unless + (Set.size (coverageFor ticks) == 2 && availableTicks ticks == 3) + (die "feedback included harness ticks or missed production ticks") + unless + (requiredModulePresent "parser" ticks && not (requiredModulePresent "lexer" ticks)) + (die "coverage-disabled stage passed its gate") + unless + (Set.null (Set.intersection (coverageFor ticks) (coverageFor changed))) + (die "different module hashes shared tick identities") + let large = BS.replicate maximumInput 255 + empty = BS.empty + unless + ( all + (\seed -> BS.length (mutate seed large large) <= maximumInput && BS.length (mutate seed empty empty) <= maximumInput) + [0 .. 10000] + ) + (die "mutation exceeded its input bound") + unless + (mutate 42 large empty == mutate 42 large empty && fingerprint empty /= fingerprint large) + (die "mutation/replay identity is inconsistent") + putStrLn "HPC feedback and mutation policy tests passed" diff --git a/Compiler/Haskell/Driver/visual-xsharp-compiler.cabal b/Compiler/Haskell/Driver/visual-xsharp-compiler.cabal index ba1444bb..47622271 100644 --- a/Compiler/Haskell/Driver/visual-xsharp-compiler.cabal +++ b/Compiler/Haskell/Driver/visual-xsharp-compiler.cabal @@ -93,6 +93,34 @@ foreign-library vxs-frontend visual-xsharp-compiler default-language: GHC2024 +executable frontend-fuzz + main-is: Main.hs + other-modules: Feedback Mutation + hs-source-dirs: Fuzzing + build-depends: + base >=4.20 && <4.23, + bytestring >=0.12 && <0.14, + containers >=0.7 && <0.9, + directory >=1.3 && <1.4, + filepath >=1.5 && <1.6, + hpc >=0.7 && <0.8, + text >=2.1 && <2.3, + visual-xsharp-compiler + ghc-options: -rtsopts "-with-rtsopts=-M512m -K8m" + default-language: GHC2024 + +test-suite fuzz-feedback-tests + type: exitcode-stdio-1.0 + main-is: Tests.hs + other-modules: Feedback Mutation + hs-source-dirs: Fuzzing + build-depends: + base >=4.20 && <4.23, + bytestring >=0.12 && <0.14, + containers >=0.7 && <0.9, + hpc >=0.7 && <0.8 + default-language: GHC2024 + test-suite visual-xsharp-compiler-tests type: exitcode-stdio-1.0 main-is: Main.hs diff --git a/Compiler/ProjectSystem/Bridge/Fuzzing/BUILD.bazel b/Compiler/ProjectSystem/Bridge/Fuzzing/BUILD.bazel new file mode 100644 index 00000000..96093643 --- /dev/null +++ b/Compiler/ProjectSystem/Bridge/Fuzzing/BUILD.bazel @@ -0,0 +1,7 @@ +load("@rules_cc//cc:cc_binary.bzl", "cc_binary") + +cc_binary( + name = "project_fuzzer", + srcs = ["ProjectFuzzer.cpp"], + deps = ["//Compiler/ProjectSystem/Bridge:project_driver", "@llvm//:llvm"], +) diff --git a/Compiler/ProjectSystem/Bridge/Fuzzing/ProjectFuzzer.cpp b/Compiler/ProjectSystem/Bridge/Fuzzing/ProjectFuzzer.cpp new file mode 100644 index 00000000..ef2a5cbf --- /dev/null +++ b/Compiler/ProjectSystem/Bridge/Fuzzing/ProjectFuzzer.cpp @@ -0,0 +1,81 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include + +#include "Compiler/ProjectSystem/Bridge/ProjectDriver.hpp" + +namespace +{ + void + Check(std::span bytes) + { + namespace Driver = Visual::XSharp::Driver; + const auto first = Driver::ParseProjectRegistry(bytes, false); + if (first != Driver::ParseProjectRegistry(bytes, false)) + llvm::report_fatal_error( + "project record decoding is nondeterministic"); + const auto required = Driver::ParseProjectRegistry(bytes, true); + if (required + && (!first + || (required->executables.empty() + && required->libraries.empty()))) + llvm::report_fatal_error( + "source-required project decoding lost its source contract"); + } +} // namespace + +extern "C" int +LLVMFuzzerTestOneInput(const std::uint8_t *data, std::size_t size) +{ + if (size > 65536U) + return 0; + Check(std::span(reinterpret_cast(data), size)); + // Reach past the version/header prefix on every input, including seeds + // that mutate one count to SIZE_MAX. No evaluator or filesystem runs here. + std::vector records{ "visual-xsharp-sources-v6", + "0.4.0", + "default", + "llvm", + "debug", + "all", + "true", + "false", + "false", + "false", + "true", + "true", + "true", + "0", + "aot", + "none", + "build/debug", + "0", + "1", + "0", + "0", + "Application", + "Example.Program", + "Sources", + "0" }; + if (size != 0U) + { + const auto selected + = static_cast(data[0]) % records.size(); + records[selected].assign(reinterpret_cast(data + 1U), + size - 1U); + } + std::string framed; + for (const auto &record : records) + { + framed.append(record); + framed.push_back('\0'); + } + Check(framed); + return 0; +} diff --git a/Compiler/ProjectSystem/Bridge/ProjectDriver.cpp b/Compiler/ProjectSystem/Bridge/ProjectDriver.cpp index 9b6c1c9f..3cf838d4 100644 --- a/Compiler/ProjectSystem/Bridge/ProjectDriver.cpp +++ b/Compiler/ProjectSystem/Bridge/ProjectDriver.cpp @@ -273,7 +273,7 @@ namespace Visual::XSharp::Driver if (!input) return std::nullopt; const auto end = static_cast(input.tellg()); - if (end <= 0 + if (end <= 0 || end > 4 * 1024 * 1024 || static_cast(end) > std::numeric_limits::max()) return std::nullopt; @@ -287,7 +287,7 @@ namespace Visual::XSharp::Driver } [[nodiscard]] std::optional> - SplitRecords(const std::vector &bytes) + SplitRecords(std::span bytes) { std::vector records; std::size_t start{}; @@ -330,7 +330,8 @@ namespace Visual::XSharp::Driver text->data() + text->size(), result); if (conversion.ec != std::errc{} - || conversion.ptr != text->data() + text->size()) + || conversion.ptr != text->data() + text->size() + || result > records_.size() - position_) return std::nullopt; return result; } @@ -441,7 +442,7 @@ namespace Visual::XSharp::Driver } [[nodiscard]] std::optional - ParseRegistry(const std::vector &bytes, bool requireSources) + ParseRegistry(std::span bytes, bool requireSources) { const auto split = SplitRecords(bytes); if (!split || split->size() < kHeaderRecordCount @@ -587,6 +588,15 @@ namespace Visual::XSharp::Driver } } // namespace + std::optional + ParseProjectRegistry(std::span bytes, bool requireSources) + { + if (bytes.empty() || bytes.size() > 4U * 1024U * 1024U + || bytes.back() != '\0') + return std::nullopt; + return ParseRegistry(bytes, requireSources); + } + std::optional ResolveProject(bool requireSources) { @@ -607,7 +617,7 @@ namespace Visual::XSharp::Driver "source registry\n"); return std::nullopt; } - auto project = ParseRegistry(*bytes, requireSources); + auto project = ParseProjectRegistry(*bytes, requireSources); if (!project) fmt::print(stderr, "vxs: bundled project evaluator returned invalid " diff --git a/Compiler/ProjectSystem/Bridge/ProjectDriver.hpp b/Compiler/ProjectSystem/Bridge/ProjectDriver.hpp index 62d5311e..02b76a46 100644 --- a/Compiler/ProjectSystem/Bridge/ProjectDriver.hpp +++ b/Compiler/ProjectSystem/Bridge/ProjectDriver.hpp @@ -5,6 +5,7 @@ #include #include +#include #include #include @@ -18,6 +19,8 @@ namespace Visual::XSharp::Driver std::optional framework; std::filesystem::path root; std::vector excludes; + bool + operator==(const ResolvedTestSuite &) const = default; }; struct ResolvedSourceTarget final @@ -28,6 +31,8 @@ namespace Visual::XSharp::Driver std::filesystem::path root; std::vector excludes; std::vector viPkgTypes; + bool + operator==(const ResolvedSourceTarget &) const = default; }; struct ResolvedProject final @@ -44,10 +49,16 @@ namespace Visual::XSharp::Driver std::filesystem::path outputDirectory; BuildOutput output{}; CompilerSettings settings{}; + bool + operator==(const ResolvedProject &) const = default; }; [[nodiscard]] std::optional ResolveProject(bool requireSources); + // Decode borrowed evaluator records without spawning a process. Every + // accepted field is copied into the returned project before bytes expire. + [[nodiscard]] std::optional + ParseProjectRegistry(std::span bytes, bool requireSources); [[nodiscard]] bool RefreshProjectLock(); } // namespace Visual::XSharp::Driver diff --git a/Compiler/ProjectSystem/Bridge/Tests/BUILD.bazel b/Compiler/ProjectSystem/Bridge/Tests/BUILD.bazel new file mode 100644 index 00000000..ea9ad500 --- /dev/null +++ b/Compiler/ProjectSystem/Bridge/Tests/BUILD.bazel @@ -0,0 +1,7 @@ +load("@rules_cc//cc:cc_binary.bzl", "cc_binary") + +cc_binary( + name = "project_registry_tests", + srcs = ["RegistryTests.cpp"], + deps = ["//Compiler/ProjectSystem/Bridge:project_driver", "@catch3//:catch2_main", "@catch3//src/Progmasoft:catch3"], +) diff --git a/Compiler/ProjectSystem/Bridge/Tests/RegistryTests.cpp b/Compiler/ProjectSystem/Bridge/Tests/RegistryTests.cpp new file mode 100644 index 00000000..9c53e4ff --- /dev/null +++ b/Compiler/ProjectSystem/Bridge/Tests/RegistryTests.cpp @@ -0,0 +1,91 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include + +#include "Compiler/ProjectSystem/Bridge/ProjectDriver.hpp" + +namespace +{ + auto + Records() -> std::vector + { + return { "visual-xsharp-sources-v6", + "0.4.0", + "default", + "llvm", + "debug", + "all", + "true", + "false", + "false", + "false", + "true", + "true", + "true", + "0", + "aot", + "none", + "build/debug", + "0", + "1", + "0", + "0", + "Application", + "Example.Program", + "Sources", + "0" }; + } + auto + Frame(const std::vector &records) -> std::string + { + std::string framed; + for (const auto &record : records) + { + framed.append(record); + framed.push_back('\0'); + } + return framed; + } +} // namespace + +TEST_CASE( + "project registry copies valid source records and rejects oversized counts") +{ + namespace Driver = Visual::XSharp::Driver; + auto records = Records(); + auto framed = Frame(records); + const auto project = Driver::ParseProjectRegistry(framed, true); + REQUIRE(project); + CHECK(project->executables.size() == 1U); + CHECK(project->entry == "Example.Program"); + framed.assign(framed.size(), 'x'); + CHECK(project->entry == "Example.Program"); + for (const auto index : { 17U, 18U, 19U, 20U, 24U }) + { + auto oversized = records; + oversized[index] = "18446744073709551615"; + CHECK_FALSE(Driver::ParseProjectRegistry(Frame(oversized), false)); + oversized[index] = "1000000"; + CHECK_FALSE(Driver::ParseProjectRegistry(Frame(oversized), false)); + } +} + +TEST_CASE("project registry validates framing and source requirements before " + "allocation") +{ + namespace Driver = Visual::XSharp::Driver; + CHECK_FALSE(Driver::ParseProjectRegistry({}, false)); + std::string oversized(4U * 1024U * 1024U + 1U, '\0'); + CHECK_FALSE(Driver::ParseProjectRegistry(oversized, false)); + auto records = Records(); + records.resize(21U); + records[18U] = "0"; + auto framed = Frame(records); + CHECK(Driver::ParseProjectRegistry(framed, false)); + CHECK_FALSE(Driver::ParseProjectRegistry(framed, true)); + framed.pop_back(); + CHECK_FALSE(Driver::ParseProjectRegistry(framed, false)); +} diff --git a/Compiler/Runtime/AARC/Fuzzing/BUILD.bazel b/Compiler/Runtime/AARC/Fuzzing/BUILD.bazel new file mode 100644 index 00000000..7aa86727 --- /dev/null +++ b/Compiler/Runtime/AARC/Fuzzing/BUILD.bazel @@ -0,0 +1,7 @@ +load("@rules_cc//cc:cc_binary.bzl", "cc_binary") + +cc_binary( + name = "ownership_fuzzer", + srcs = ["OwnershipFuzzer.cpp"], + deps = ["//Compiler/Runtime/AARC:aarc", "@llvm//:llvm"], +) diff --git a/Compiler/Runtime/AARC/Fuzzing/OwnershipFuzzer.cpp b/Compiler/Runtime/AARC/Fuzzing/OwnershipFuzzer.cpp new file mode 100644 index 00000000..efbdf96c --- /dev/null +++ b/Compiler/Runtime/AARC/Fuzzing/OwnershipFuzzer.cpp @@ -0,0 +1,100 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Runtime/AARC.hpp" + +namespace +{ + namespace Aarc = Visual::XSharp::Runtime::Aarc; + constexpr std::uint64_t kMarker = 0xABCDEF123456ULL; + struct Payload final + { + std::atomic_uint32_t *destructions{}; + std::uint64_t marker{}; + }; + void + Destroy(void *object) noexcept + { + auto *payload = static_cast(object); + payload->marker = 0U; + payload->destructions->fetch_add(1U, std::memory_order_relaxed); + } + const Aarc::TypeMetadata kMetadata{ Aarc::kAbiVersion, + 0U, + Aarc::TypeIdentity("Fuzz.Payload"), + sizeof(Payload), + alignof(Payload), + Destroy, + "Fuzz.Payload" }; +} // namespace + +extern "C" int +LLVMFuzzerTestOneInput(const std::uint8_t *data, std::size_t size) +{ + std::atomic_uint32_t destructions{}; + auto *payload = static_cast(Aarc::Allocate(kMetadata)); + if (payload == nullptr) + llvm::report_fatal_error("ownership fuzz payload allocation failed"); + payload->destructions = &destructions; + payload->marker = kMarker; + const auto weak = Aarc::MakeWeak(payload); + const auto unowned = Aarc::MakeUnowned(payload); + std::atomic_bool start{}; + std::atomic_bool corrupt{}; + const auto workerCount + = size == 0U ? 2U : 1U + static_cast(data[0] % 4U); + const auto iterations + = size < 2U ? 8U : 1U + static_cast(data[1] % 64U); + // std::jthread is unavailable in Apple libc++; join explicitly below. + std::vector workers; + workers.reserve(workerCount); + for (unsigned worker = 0U; worker < workerCount; ++worker) + { + const auto localWeak = Aarc::CopyWeak(weak); + const auto localUnowned = Aarc::CopyUnowned(unowned); + workers.emplace_back([&, localWeak, localUnowned, worker] { + while (!start.load(std::memory_order_acquire)) + std::this_thread::yield(); + for (unsigned iteration = 0U; iteration < iterations; ++iteration) + { + auto *locked = static_cast( + (iteration + worker) % 2U == 0U + ? Aarc::LockWeak(localWeak) + : Aarc::LoadUnowned(localUnowned)); + if (locked != nullptr) + { + // Reads occur only while a temporary strong owner protects + // the payload. Destructor writes must never overlap them. + if (locked->marker != kMarker) + corrupt.store(true, std::memory_order_relaxed); + Aarc::ReleaseStrong(locked); + } + const auto extra = Aarc::CopyWeak(localWeak); + Aarc::ReleaseWeak(extra); + std::this_thread::yield(); + } + Aarc::ReleaseWeak(localWeak); + Aarc::ReleaseUnowned(localUnowned); + }); + } + start.store(true, std::memory_order_release); + Aarc::ReleaseStrong(payload); + // Every worker finishes before the independent destructor oracle runs. + for (auto &worker : workers) + worker.join(); + if (corrupt.load() || destructions.load() != 1U + || Aarc::LockWeak(weak) != nullptr + || Aarc::LoadUnowned(unowned) != nullptr) + llvm::report_fatal_error( + "ownership race resurrected or multiply destroyed a payload"); + Aarc::ReleaseWeak(weak); + Aarc::ReleaseUnowned(unowned); + return 0; +} diff --git a/Documents/CORE-IR.md b/Documents/CORE-IR.md index 3ac3916c..f719d92e 100644 --- a/Documents/CORE-IR.md +++ b/Documents/CORE-IR.md @@ -241,6 +241,45 @@ moving evaluation across a `break`, `continue`, or return. CorePrep consumes the verified structure and materializes loop headers, exits, latches, and the distinct `for` update block. +The Haskell CorePrep lowering and the native Core-to-CorePrep adapter must +build the same loop shape: + +- a `while` or `for` condition owns a dedicated header block. The statements + that precede the loop stay in the incoming block, which jumps to the header + once; every back-edge targets the header, never the incoming block; +- a numeric condition's canonicalizing `value != 0` comparison belongs to that + header and is re-evaluated on every iteration; +- the `for` update region is entered by normal body completion and by + `continue`, and every open tail of the update region jumps to the header. + The region's own entry is its `continue` target but is not its successor; +- `&&` and `||` are control flow, not eager two-operand instructions. A + Boolean result slot is initialized with the short-circuit value, the left + operand selects a branch, and only the block on the evaluating edge computes + the right operand and overwrites the slot. Calls, traps, and non-termination + in the right operand therefore stay conditional. A short-circuit condition + of a loop is evaluated starting at the loop header, so the loop's own + body/exit branch may sit in the operator's join block. + +`LoopLoweringTests.cpp` and `ShortCircuitLoweringTests.cpp` under +`Compiler/Core/Tests/` assert these edges exactly on the native adapter. +`LoopExecutionTests.cpp` and `ShortCircuitExecutionTests.cpp` under +`Compiler/Backend/LLVM/Tests/` execute each form through CorePrep, Xpp, Xmm, +and LLVM with both native optimizer settings and compare the result with host +code; the short-circuit programs guard a division or a recursive call, so an +eager right operand traps or never returns instead of merely producing the +same Boolean. A CorePrep, Xpp, +or Xmm verifier cannot reject a wrong back-edge by itself: a block that jumps +to itself is a well-formed control-flow graph. + +Both lowerings also allocate generated symbols the same way. Temporaries, +condition and short-circuit slots, and lifted closure names come from one +counter that starts above every symbol identity in the whole module and is +never reset between functions. Identities are module-wide and CorePrep +verification rejects one identity with two spellings, so a counter seeded +from a single function would reuse another function's symbols. +`SymbolAllocationTests.cpp` covers this for multi-function modules and +closures. + ## Expressions Every Core expression has a statically queryable type. diff --git a/Documents/DEVELOPER-RECIPES.md b/Documents/DEVELOPER-RECIPES.md index dc47afc6..63f7e604 100644 --- a/Documents/DEVELOPER-RECIPES.md +++ b/Documents/DEVELOPER-RECIPES.md @@ -17,6 +17,7 @@ just test just benchmark just fuzz just fuzz-stress +just fuzz-thread # TSan run of threaded fuzz targets on a supported host just sanitize # ASan + UBSan just sanitize-thread # Separate TSan run on a supported host just verify-helpers diff --git a/Documents/FUZZING.md b/Documents/FUZZING.md index a9c22acf..81d6075f 100644 --- a/Documents/FUZZING.md +++ b/Documents/FUZZING.md @@ -14,20 +14,53 @@ memory safety or complete language coverage. | `wire_fuzzer` | Core, private CorePrep transport, Xpp and Xmm; bounded decoding, semantic verification and equal encode/decode round trips | First-party C++ codecs and verifiers | | `lexer_fuzzer` | Arbitrary bytes through the frontend lexer ABI; complete token/diagnostic evaluation | Native ABI bridge, not GHC-generated lexer branches | | `parser_fuzzer` | Arbitrary bytes through syntax analysis; complete AST/diagnostic evaluation | Native ABI bridge, not GHC-generated parser branches | -| `source_llvm_fuzzer` | Source through Core/CorePrep, Xpp/Xmm verification and LLVM lowering; generated arithmetic differential oracle | First-party C++ pipeline and JIT bridge | - -The source oracle independently evaluates bounded generated arithmetic, compiles -it with Xpp/Xmm optimizations both disabled and enabled, executes both verified -artifacts through ORC, and compares all three results. This detects miscompiles -within that generated subset; it is not an oracle for arbitrary Visual X# +| `source_llvm_fuzzer` | Arbitrary source through Core/CorePrep, Xpp/Xmm verification and LLVM lowering | First-party C++ pipeline | +| `differential_fuzzer` | Generated arithmetic and control flow compiled with native optimizers disabled/enabled and compared with an independent evaluator | First-party C++ pipeline and JIT bridge | +| `cli_fuzzer` | NUL-separated arguments through the typed command-line parser; two parses must produce equal typed models, leave `argv` unchanged, attach a diagnostic to every rejection and select a command on every acceptance | First-party C++ CLI parser | +| `project_fuzzer` | Evaluator registry records, both raw and as one mutated field of a valid document; decoding must be deterministic and a source-requiring decode may only accept what the permissive decode accepts | First-party C++ registry decoder; no evaluator process or filesystem access | +| `repl_fuzzer` | Up to eight operations on one persistent session: arithmetic cells checked against an independent accumulator, type queries that must not change history, failed-cell rollback and reset | First-party C++ session and JIT bridge | +| `ownership_fuzzer` | One to four threads racing weak locks, unowned loads and weak copies against the final strong release; the payload must be destroyed exactly once, never resurrected and never read after destruction | AARC runtime | +| `frontend-fuzz` | Lexer, parser and source-to-CorePrep stages mutated in-process with GHC HPC tick feedback | GHC-compiled frontend modules; no native sanitizer | + +The differential oracle independently evaluates one generated program per +input: a bounded arithmetic expression, a classic `for` loop with `continue` +and `break`, a `do`/`while` loop, an `if`/`else` over that expression, a +`while` loop preceded by its own initializers, or a recursion that terminates +only because `||` and `&&` skip their right operands. The expected value is computed +by ordinary host code in the harness, never by a second compiler path. It +compiles the source once, lowers the same verified Core with Xpp/Xmm +optimizations both disabled and enabled, executes both verified +artifacts through ORC, and compares all three results. Agreement between the +two optimizer settings is not accepted by itself: both consume the same +Core-to-CorePrep adapter, so a defect there makes them agree on the same wrong +or non-terminating program. This detects miscompiles within that generated +subset; it is not an oracle for arbitrary Visual X# programs. Invalid source is a normal rejection, whereas internal failures and verified-model inconsistencies fail the campaign. +Arbitrary source and generated arithmetic use separate corpora and equal +per-target time budgets. This lets source mutations reach native lowering +without repeatedly creating two ORC sessions for unrelated generated code. +The differential generator consumes at most 33 bytes: 31 expression selectors, +one shape byte and one trip-count byte. Its 64-byte input limit keeps +mutations near the bytes that influence the program. + GHC frontend code and prebuilt LLVM dependencies do not receive Clang native -coverage instrumentation. Lexer/parser execution must not be presented as -coverage-guided exploration of their Haskell implementation. Haskell tests and -HPC coverage remain complementary gates, not substitutes for a frontend-specific -feedback-guided engine. +coverage instrumentation. Running `lexer_fuzzer` or `parser_fuzzer` must not be +presented as coverage-guided exploration of the Haskell implementation. + +`frontend-fuzz` is the separate feedback engine for that code. It is a Cabal +executable built with `--enable-coverage` in its own `dist-fuzz-coverage` +build directory, so it never shares package state with the ordinary frontend +build. It resets the HPC counters before each input, keeps an input when it +reaches a tick of a `Visual.XSharp.*` module that no earlier input reached, +and refuses to run when the production modules carry no ticks. Mutations are +deterministic for a recorded seed, inputs are limited to 8192 bytes and five +seconds, and the runtime heap is limited to 512 MiB. A Haskell exception or a +timeout writes the exact input as `failure.seed` and fails the campaign; +`frontend-fuzz STAGE --replay FILE` replays it. This engine has no sanitizer +and reports tick coverage, which is not comparable with libFuzzer edge +coverage. ## Run a campaign @@ -46,6 +79,28 @@ target; `fuzz-stress` defaults to 900. Set `VXS_FUZZ_SECONDS` to an integer from 1 through 3600 to override either duration. CI uses 90 seconds per target for bounded campaigns and 900 for scheduled stress campaigns. +Targets are independent processes with their own corpus, artifact directory +and report entry, so the helper can run several at once. Concurrency is not +free evidence: a campaign is worth the inputs it executes inside its time +budget, and on a two-core, four-thread host two concurrent targets executed +40 to 90 percent fewer inputs each. The default is therefore one job per four +logical processors, at most four, which is one target at a time on such a host +and on four-vCPU CI runners. Set `VXS_FUZZ_JOBS` to an integer from 1 through +64 on a host with spare cores. Two targets with a 4096 MiB RSS limit never +overlap, the three HPC stages follow the same setting, and the ThreadSanitizer +campaign always runs one target at a time because its targets start their own +threads. Time budgets, RSS limits, per-input timeouts and watchdogs are +identical at every job count. + +Most local wall-clock time is compilation, not fuzzing. The plain, sanitizer +and fuzz configurations share one Bazel output tree, and each switch would +otherwise recompile every owned translation unit. Outside CI the helper adds a +persistent content-addressed Bazel disk cache under the user cache directory +(`visual-xsharp/bazel-disk-cache`, limited to 8 GiB). It is keyed by each +action's full command line and inputs, so no instrumentation or check changes. +`VXS_BAZEL_DISK_CACHE` selects another absolute directory, or `off` for a cold +measurement. + Each campaign has a 30-second per-input timeout and a finite input length: | Target | Maximum input bytes | RSS limit (MiB) | @@ -54,6 +109,11 @@ Each campaign has a 30-second per-input timeout and a finite input length: | Lexer | 65536 | 1024 | | Parser | 65536 | 1536 | | Source/LLVM | 65536 | 4096 | +| Differential | 64 | 4096 | +| CLI | 16384 | 768 | +| Project registry | 65536 | 768 | +| REPL | 64 | 4096 | +| Ownership | 64 | 768 | ASan intentionally retains freed allocations in quarantine. Fuzz-only settings bound this cache to 64 MiB, with a 256 KiB thread-local cache; the nonzero @@ -64,15 +124,20 @@ normal quarantine settings. ## Corpus synchronization and reports Wire seeds come from production writers, so format-version changes do not leave -handwritten supposedly valid documents behind. Lexer, parser and source seeds -come from `Compiler/Fuzzing/Corpus/`. Set `VXS_FUZZ_CORPUS` to retain mutation +handwritten supposedly valid documents behind. Every other target has +versioned seeds under `Compiler/Fuzzing/Corpus//`; a target without +that directory fails the campaign instead of starting from nothing. Set `VXS_FUZZ_CORPUS` to retain mutation corpora across local runs. Updated versioned seeds are added without overwriting older discovered inputs; conflicting contents under a stable hash fail closed. GitHub Actions restores a per-platform corpus cache and saves a unique cache version for each run. It also uploads campaign artifacts on success or failure. Reports contain the target, duration, RSS limit, selected sanitizer, native -coverage ownership and result. Logs include execution counts and peak RSS. +coverage ownership and result. Structured reports also include executed inputs, +average executions per second, new corpus entries, slowest input time and peak +RSS from libFuzzer's final counters. A successful process without a complete +final report or with zero executed inputs fails the campaign gate. Failed +processes retain their original logs even if final counters are unavailable. Failed work directories are preserved for diagnosis; successful CI reports are retained for artifact upload. @@ -93,8 +158,31 @@ whole corpus and sanitizer cache rather than just the final input. `wire_fuzz_smoke` exercises 1024 deterministic mutations of each of four valid documents. It runs independently of libFuzzer and does not claim guided -coverage. `source_fuzz_smoke` checks valid-source lowering and the differential -oracle before mutation campaigns begin. +coverage. `source_fuzz_smoke` checks valid-source lowering and then runs the +differential oracle on every generated program shape at trip counts 0 through +11 before mutation campaigns begin. Set `VXS_FUZZ_TRACE=1` to print each +generated source with its reference and optimized LLVM IR. + +Neither smoke program nor the HPC engine has libFuzzer's per-input timeout, and +a miscompiled generated loop does not return. The developer helper therefore +bounds every smoke, libFuzzer and HPC process as a whole: 90 seconds for a +smoke program, the campaign duration plus 90 seconds for a libFuzzer target and +plus 300 seconds for the HPC engine. On expiry it terminates the complete +process tree and reports a watchdog failure. Build tools are not bounded by +this watchdog. Do not run a smoke binary directly without an external time +limit when investigating a suspected hang. + +## ThreadSanitizer campaign + +AddressSanitizer and ThreadSanitizer cannot share one executable. +`go run ./helpers/cmd/develop fuzz-thread` builds only the targets that start +their own threads, currently `ownership_fuzzer`, with libFuzzer and +ThreadSanitizer, proves that the runtime reports an intentional data race, and +then runs a bounded campaign with the same corpus, limits and report checks. +The command fails on Windows, where Clang has no ThreadSanitizer runtime; it +never reports a skipped run as success. The fuzzing workflow runs it on macOS +and native Ubuntu. The Fedora container job does not run it for the +shadow-memory reason given below. Compiler Tier 1/2/3 run complete native suites with ASan/UBSan. Tier 1 macOS and Tier 2 native Ubuntu additionally run separate TSan suites. Fedora Tier 3 @@ -110,8 +198,22 @@ protection must require these actual checks. ## Remaining coverage boundaries -CLI argument generation, project/lockfile inputs, persistent REPL sessions and -broader generated language programs need independent oracles. Deep semantic -cases, ownership concurrency and frontend feedback-guided coverage are not -established by the four existing targets. Expand these deliberately instead of -equating a green workflow with completion of the entire security program. +The harnesses above are evidence about the inputs they ran, not proofs: + +- the CLI, project and REPL oracles check determinism, rollback and a small + arithmetic model. They do not model every option interaction, lockfile + refresh, project evaluation or REPL declaration form; +- the differential generator covers integer arithmetic, three loop forms, + one conditional and one guarded recursion. Closures, other scalar types, + ownership and templates have no generated-program oracle here; their + executable checks live in the component test suites; +- `ownership_fuzzer` varies thread count and iteration count, not arbitrary + interleavings, and its ThreadSanitizer campaign does not run on Windows or in + the Fedora container; +- HPC feedback is expression-tick coverage of the frontend, without + memory-safety instrumentation, and its mutator is byte- and token-level, not + grammar-aware; +- prebuilt LLVM and the GHC runtime are not instrumented by any campaign. + +Expand these deliberately instead of equating a green workflow with completion +of the entire security program. diff --git a/Documents/INTERACTIVE.md b/Documents/INTERACTIVE.md index c1054c32..8e8793c6 100644 --- a/Documents/INTERACTIVE.md +++ b/Documents/INTERACTIVE.md @@ -155,6 +155,13 @@ JIT execution runs in-process. As with any compiler REPL, evaluating untrusted p with the current user's authority. `vxsi` intentionally does not claim to provide a sandbox, process isolation, memory quota, or security boundary. +On Windows the ORC session links each object into one contiguous memory reservation. Win64 unwind tables use +image-relative relocations that the runtime linker can apply only when no section of an object lies below the lowest one; +with separately mapped sections the operating system chooses that order, and some address layouts abort linking with +"relocation requires an ordered section layout". The failure is layout dependent and was observed once in a hosted +AddressSanitizer run; it has not been reproduced locally, so the reservation removes the cause rather than a +demonstrated repeatable failure. Other hosts keep LLJIT's default linking layer. + ## Build and verify The native targets belong to the repository root, matching the feature's ownership tree: diff --git a/Documents/LLVM-BACKEND.md b/Documents/LLVM-BACKEND.md index 240cbe4d..3766c811 100644 --- a/Documents/LLVM-BACKEND.md +++ b/Documents/LLVM-BACKEND.md @@ -61,6 +61,15 @@ verifier require one condition shape. ## Calls and entry bridge +A function's LLVM symbol is its dotted module name, its source spelling, and +its symbol identity, for example `Demo.Main.7`. LLVM reserves every +global name that begins with `llvm.` for intrinsics and rejects a module that +defines one, while `llvm` is an ordinary namespace name in source. Such a +symbol is emitted with a leading `$`, which cannot occur in a source +identifier, for example `$llvm.Main.7`. Declarations in other +per-source objects use the same rule, so cross-object references still +resolve. + Direct calls resolve a stable function symbol to a declared function. Parameter count/types and result type are checked in Xmm before LLVM call construction. A function symbol is not encoded as an integer or ordinary virtual register. diff --git a/Documents/TESTING.md b/Documents/TESTING.md index 2b566d1b..f35a3205 100644 --- a/Documents/TESTING.md +++ b/Documents/TESTING.md @@ -118,7 +118,8 @@ go run ./helpers/cmd/develop doctor go run ./helpers/cmd/develop test ``` -The command executes 18 Catch3 binaries and one C11 ABI contract executable on +The command executes 19 Catch3 binaries, one C11 ABI contract executable and +one source/differential fuzz smoke executable on Windows 10/11, macOS Sequoia/Tahoe, Ubuntu 26.04 LTS, and Fedora 43. Bazel selects the host configuration automatically; no public test instruction requires `--config`. @@ -142,14 +143,18 @@ labels shorten iteration, but they do not replace the full native gate. See Run native memory diagnostics through the same entry point: ```powershell -go run ./helpers/cmd/develop sanitize address +go run ./helpers/cmd/develop sanitize address-undefined ``` -macOS and Linux additionally support `sanitize undefined` and `sanitize thread`. Each sanitizer instruments both compilation and -linking and runs the complete native suite set rather than merely proving that instrumented objects compile. +Windows, macOS and Linux support combined ASan/UBSan and their individual +`sanitize address` / `sanitize undefined` profiles. macOS and native Linux also +support the separate `sanitize thread` profile. Each profile verifies real +intentional failures before running the complete native suite set. See +[Fuzzing](FUZZING.md) for the Windows and Fedora TSan boundaries. -The Compiler Tier 1/2/3 workflows also run a separate deterministic wire-mutation smoke target. It is not included in the 16 -component-owned `develop.go test` suites. Run it directly when a Core, CorePrep, Xpp, or Xmm decoder changes: +The Compiler Tier 1/2/3 workflows also run a separate deterministic wire-mutation +smoke target. It is separate from the component-owned `develop test` suite set. +Run it directly when a Core, CorePrep, Xpp, or Xmm decoder changes: ```powershell bazelisk build //Compiler/Fuzzing:wire_fuzz_smoke @@ -336,7 +341,7 @@ or below 1500 lines. A simple review aid is: ```powershell $extensions = '*.hs','*.cpp','*.hpp','*.hh','*.kt','*.kts','*.go','*.java' -Get-ChildItem Compiler,Interactive,ProjectSystem,Analyzer,Formatter,Linter,scripts -Recurse -File -Include $extensions | +Get-ChildItem Compiler,Interactive,ProjectSystem,Analyzer,Formatter,Linter,helpers -Recurse -File -Include $extensions | Where-Object { (Get-Content -LiteralPath $_.FullName).Count -gt 1500 } | Select-Object FullName ``` diff --git a/Interactive/Fuzzing/BUILD.bazel b/Interactive/Fuzzing/BUILD.bazel new file mode 100644 index 00000000..ac6de162 --- /dev/null +++ b/Interactive/Fuzzing/BUILD.bazel @@ -0,0 +1,7 @@ +load("@rules_cc//cc:cc_binary.bzl", "cc_binary") + +cc_binary( + name = "repl_fuzzer", + srcs = ["ReplFuzzer.cpp"], + deps = ["//Interactive:interactive_runtime", "@llvm//:llvm"], +) diff --git a/Interactive/Fuzzing/ReplFuzzer.cpp b/Interactive/Fuzzing/ReplFuzzer.cpp new file mode 100644 index 00000000..8bb49415 --- /dev/null +++ b/Interactive/Fuzzing/ReplFuzzer.cpp @@ -0,0 +1,70 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Interactive/Session.hpp" + +extern "C" int +LLVMFuzzerTestOneInput(const std::uint8_t *data, std::size_t size) +{ + namespace Repl = Visual::XSharp::Interactive; + Repl::Session session; + std::int64_t expected{}; + bool hasPrevious{}; + const auto initial = session.Evaluate("return 0;"); + if (initial.status != Repl::CellStatus::Value) + llvm::report_fatal_error( + "REPL fuzz runtime could not evaluate its valid initial cell"); + hasPrevious = true; + for (std::size_t index = 0U; index < std::min(size, std::size_t{ 8U }); + ++index) + { + const auto selector = data[index]; + if (selector % 5U == 0U) + { + if (session.Reset() || !session.History().empty()) + llvm::report_fatal_error( + "REPL reset retained failed resources or history"); + hasPrevious = false; + expected = 0; + continue; + } + if (selector % 5U == 1U) + { + const auto history = session.History(); + if (session.Evaluate("return MissingFuzzName;").status + != Repl::CellStatus::Error + || session.History() != history) + llvm::report_fatal_error( + "REPL failed-cell rollback changed history"); + continue; + } + const auto increment = static_cast(selector % 11U); + const auto expression = std::string("return ") + + (hasPrevious ? "vxsiPrevious + " : "") + + std::to_string(increment) + ";"; + const auto history = session.History(); + if (session.TypeOf(expression).status != Repl::CellStatus::Type + || session.History() != history) + llvm::report_fatal_error("REPL type query mutated session state"); + expected += increment; + const auto value = session.Evaluate(expression); + const auto *integer + = value.value ? std::get_if(&value.value->payload) + : nullptr; + if (value.status != Repl::CellStatus::Value || integer == nullptr + || *integer != expected) + llvm::report_fatal_error("persistent REPL session disagrees with " + "independent arithmetic state"); + hasPrevious = true; + } + if (session.Reset()) + llvm::report_fatal_error("REPL fuzz resource teardown failed"); + return 0; +} diff --git a/REUSE.toml b/REUSE.toml index 05f4ce9e..79fcb978 100644 --- a/REUSE.toml +++ b/REUSE.toml @@ -8,3 +8,9 @@ path = [".bazelversion", "MODULE.bazel.lock", ".github/brand/visual-xsharp-socia precedence = "aggregate" SPDX-FileCopyrightText = "2026 Progmasoft " SPDX-License-Identifier = "MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1" + +[[annotations]] +path = ["Compiler/Fuzzing/Corpus/**"] +precedence = "aggregate" +SPDX-FileCopyrightText = "2026 Progmasoft " +SPDX-License-Identifier = "MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1" diff --git a/helpers/internal/development/bazel_cache.go b/helpers/internal/development/bazel_cache.go new file mode 100644 index 00000000..d5c052ff --- /dev/null +++ b/helpers/internal/development/bazel_cache.go @@ -0,0 +1,48 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "os" + "path/filepath" +) + +// bazelDiskCacheLimit bounds the local cache; Bazel evicts the least recently +// used entries beyond it while idle. +const bazelDiskCacheLimit = "8G" + +// cachedBuild adds a persistent, content-addressed disk cache to a Bazel build. +// One output tree serves the plain, sanitizer and fuzz configurations in turn, +// and Bazel's in-tree action cache remembers only the most recent action per +// output, so every configuration switch otherwise recompiles all owned +// translation units even when no source changed. The disk cache is keyed by +// each action's complete command line and inputs: a sanitizer or fuzz object +// can never be served to another configuration, and no instrumentation, +// check or time budget is altered. +// +// CI keeps its own cache through the workflow's Bazel setup, and +// VXS_BAZEL_DISK_CACHE=off disables this one for a cold local measurement. +func cachedBuild(arguments []string) []string { + directory, enabled := bazelDiskCacheDirectory(os.Getenv("CI"), os.Getenv("VXS_BAZEL_DISK_CACHE"), os.UserCacheDir) + if !enabled || len(arguments) == 0 || arguments[0] != "build" { + return arguments + } + cached := make([]string, 0, len(arguments)+2) + cached = append(cached, "build", "--disk_cache="+directory, "--experimental_disk_cache_gc_max_size="+bazelDiskCacheLimit) + return append(cached, arguments[1:]...) +} + +func bazelDiskCacheDirectory(ci, configured string, userCache func() (string, error)) (string, bool) { + if ci == "true" || configured == "off" { + return "", false + } + if configured != "" { + return configured, filepath.IsAbs(configured) + } + root, err := userCache() + if err != nil || root == "" { + return "", false + } + return filepath.Join(root, "visual-xsharp", "bazel-disk-cache"), true +} diff --git a/helpers/internal/development/bazel_cache_test.go b/helpers/internal/development/bazel_cache_test.go new file mode 100644 index 00000000..e97030eb --- /dev/null +++ b/helpers/internal/development/bazel_cache_test.go @@ -0,0 +1,51 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "errors" + "path/filepath" + "testing" +) + +func TestBazelDiskCacheIsLocalOptionalAndAbsolute(t *testing.T) { + root := t.TempDir() + cache := func() (string, error) { return root, nil } + if directory, enabled := bazelDiskCacheDirectory("", "", cache); !enabled || directory != filepath.Join(root, "visual-xsharp", "bazel-disk-cache") { + t.Fatalf("default cache: %q, %v", directory, enabled) + } + // CI owns its cache through the workflow; a developer may opt out. + for _, example := range [][2]string{{"true", ""}, {"", "off"}, {"true", root}} { + if directory, enabled := bazelDiskCacheDirectory(example[0], example[1], cache); enabled { + t.Fatalf("cache enabled for CI=%q setting=%q: %q", example[0], example[1], directory) + } + } + if directory, enabled := bazelDiskCacheDirectory("", root, cache); !enabled || directory != root { + t.Fatalf("explicit cache: %q, %v", directory, enabled) + } + // A relative path would resolve inside Bazel's own working directory. + if _, enabled := bazelDiskCacheDirectory("", "relative/cache", cache); enabled { + t.Fatal("relative cache directory accepted") + } + if _, enabled := bazelDiskCacheDirectory("", "", func() (string, error) { return "", errors.New("no cache directory") }); enabled { + t.Fatal("cache enabled without a user cache directory") + } +} + +func TestCachedBuildOnlyExtendsBuildCommands(t *testing.T) { + t.Setenv("CI", "") + t.Setenv("VXS_BAZEL_DISK_CACHE", t.TempDir()) + build := cachedBuild([]string{"build", "--config=asan-linux", "//Compiler/Cli:vxs"}) + if len(build) != 5 || build[0] != "build" || build[3] != "--config=asan-linux" || build[4] != "//Compiler/Cli:vxs" { + t.Fatalf("build arguments were reordered or dropped: %v", build) + } + clean := []string{"clean", "--expunge"} + if got := cachedBuild(clean); len(got) != 2 { + t.Fatalf("non-build command changed: %v", got) + } + t.Setenv("VXS_BAZEL_DISK_CACHE", "off") + if got := cachedBuild([]string{"build", "//Compiler/Cli:vxs"}); len(got) != 2 { + t.Fatalf("disabled cache still added options: %v", got) + } +} diff --git a/helpers/internal/development/build.go b/helpers/internal/development/build.go index e12a77a0..2750aec6 100644 --- a/helpers/internal/development/build.go +++ b/helpers/internal/development/build.go @@ -33,6 +33,7 @@ var nativeTargets = []string{ "//Compiler/Runtime/AARC/Tests:aarc_runtime_tests", "//Compiler/Runtime/AARC/Tests:aarc_c_abi_tests", "//Compiler/Fuzzing:source_fuzz_smoke", + "//Compiler/ProjectSystem/Bridge/Tests:project_registry_tests", "//Interactive/Tests:interactive_tests", } @@ -78,7 +79,7 @@ func buildTargets(repository string, runner commandRunner, config string, extra arguments = append(arguments, nativeTargets...) arguments = append(arguments, extra...) fmt.Printf("Building compiler and %d native suites...\n", len(nativeTargets)) - if err := runner.Run(repository, nil, bazel, arguments...); err != nil { + if err := runner.Run(repository, nil, bazel, cachedBuild(arguments)...); err != nil { return fmt.Errorf("Bazel build failed: %w", err) } if err := stageFrontendForBuildOutputs(repository, frontendLibrary); err != nil { @@ -183,7 +184,7 @@ func runBenchmarks(repository string, currentHost host, runner commandRunner, ba arguments := append([]string{"build", "-c", "opt"}, nativeBenchmarkTargets...) arguments = append(arguments, bazelArguments...) fmt.Printf("Building %d native benchmark programs...\n", len(nativeBenchmarkTargets)) - if err := runner.Run(repository, nil, bazel, arguments...); err != nil { + if err := runner.Run(repository, nil, bazel, cachedBuild(arguments)...); err != nil { return fmt.Errorf("native benchmark build failed: %w", err) } for index, program := range nativeBenchmarkPrograms { diff --git a/helpers/internal/development/clean.go b/helpers/internal/development/clean.go index f8ba7462..aee2023d 100644 --- a/helpers/internal/development/clean.go +++ b/helpers/internal/development/clean.go @@ -12,6 +12,7 @@ import ( var generatedBuildPaths = []string{ "Compiler/dist-newstyle", + "Compiler/dist-fuzz-coverage", "ProjectSystem/.gradle", "ProjectSystem/build", "Analyzer/.gradle", diff --git a/helpers/internal/development/cli.go b/helpers/internal/development/cli.go index d9b1b398..edccb5b2 100644 --- a/helpers/internal/development/cli.go +++ b/helpers/internal/development/cli.go @@ -39,7 +39,8 @@ func newCommand(runner commandRunner, output, errorOutput io.Writer) *cobra.Comm {"test", "Build and execute every native contract suite.", 0, true}, {"sanitize", "Run address, undefined, address-undefined, or thread sanitizer suites.", 1, true}, {"version", "Validate major.minor.patch[.revision] release metadata.", 1, false}, - {"fuzz", "Run short wire, lexer, parser, and LLVM-source campaigns.", 0, false}, + {"fuzz", "Run bounded ASan/UBSan libFuzzer campaigns and the Haskell HPC campaign.", 0, false}, + {"fuzz-thread", "Run the threaded fuzz targets under ThreadSanitizer (macOS/Linux).", 0, false}, {"incremental-clean-build", "Rebuild while preserving downloaded dependencies.", 0, false}, {"cold-clean-build", "Expunge Bazel state and rebuild the compiler.", 0, false}, {"clean", "Remove generated Bazel, Cabal, and Gradle output.", 0, false}, diff --git a/helpers/internal/development/cli_test.go b/helpers/internal/development/cli_test.go index 6eb14b6f..95ea1202 100644 --- a/helpers/internal/development/cli_test.go +++ b/helpers/internal/development/cli_test.go @@ -60,12 +60,12 @@ func TestFuzzBuildPlansSeparateSmokeAndRuntimeMain(t *testing.T) { } drivers := 0 for _, argument := range campaign { - if strings.HasPrefix(argument, "//Compiler/Fuzzing:") { + if strings.HasPrefix(argument, "//") { drivers++ } } - if drivers != 4 { - t.Fatalf("expected four campaign drivers: %v", campaign) + if drivers != 9 { + t.Fatalf("expected nine campaign drivers: %v", campaign) } if sanitizer != "" { for _, plan := range [][]string{smoke, campaign} { diff --git a/helpers/internal/development/commands.go b/helpers/internal/development/commands.go index 523ae4f8..f8bc7e77 100644 --- a/helpers/internal/development/commands.go +++ b/helpers/internal/development/commands.go @@ -78,6 +78,14 @@ func executeWorkflow(arguments []string, runner commandRunner) error { return err } return runFuzzCampaign(repository, currentHost, runner, false, true) + case "fuzz-thread": + if len(commandArguments) != 0 || len(bazelArguments) != 0 { + return errors.New("fuzz-thread does not accept arguments") + } + if err := requireBuildTools(currentHost, runner); err != nil { + return err + } + return runThreadFuzzCampaign(repository, currentHost, runner) case "fuzz-stress": if len(bazelArguments) != 0 || (len(commandArguments) != 0 && !(len(commandArguments) == 1 && strings.EqualFold(commandArguments[0], "--asan"))) { return errors.New("fuzz-stress accepts only the optional --asan flag") diff --git a/helpers/internal/development/fuzz.go b/helpers/internal/development/fuzz.go index 2c33cc10..3c2cc46a 100644 --- a/helpers/internal/development/fuzz.go +++ b/helpers/internal/development/fuzz.go @@ -13,6 +13,7 @@ import ( "path/filepath" "strconv" "strings" + "sync" ) func fuzzConfiguration(currentHost host) (string, error) { @@ -39,7 +40,7 @@ func macOSFuzzerRuntime(root string) (string, error) { return matches[0], nil } -// Smoke programs own main; only the four campaign drivers may link libFuzzer's +// Smoke programs own main; only campaign drivers may link libFuzzer's // main. Sharing one global fuzz profile with both groups duplicates main on // Linux, where -fsanitize=fuzzer pulls the driver in unconditionally. func fuzzBuildArguments(configuration, sanitizerConfiguration, macRuntime string) ([]string, []string) { @@ -53,11 +54,9 @@ func fuzzBuildArguments(configuration, sanitizerConfiguration, macRuntime string campaign = append(campaign, "--linkopt="+macRuntime) } smoke = append(smoke, "//Compiler/Fuzzing:wire_fuzz_smoke", "//Compiler/Fuzzing:source_fuzz_smoke") - campaign = append(campaign, - "//Compiler/Fuzzing:wire_fuzzer", - "//Compiler/Fuzzing:lexer_fuzzer", - "//Compiler/Fuzzing:parser_fuzzer", - "//Compiler/Fuzzing:source_llvm_fuzzer") + for _, target := range nativeFuzzTargets() { + campaign = append(campaign, target.label) + } return smoke, campaign } @@ -108,8 +107,8 @@ func runFuzzCampaign(repository string, currentHost host, runner commandRunner, if err := os.Mkdir(artifacts, 0o700); err != nil { return fmt.Errorf("could not create fuzz artifact directory %q: %w", work, err) } - stageNames := []string{"wire", "lexer", "parser", "source"} - for _, stage := range stageNames { + for _, target := range nativeFuzzTargets() { + stage := target.corpus corpus := filepath.Join(corpusRoot, stage) if err := os.MkdirAll(corpus, 0o700); err != nil { return fmt.Errorf("could not prepare %s corpus; preserved %q: %w", stage, work, err) @@ -119,11 +118,7 @@ func runFuzzCampaign(repository string, currentHost host, runner commandRunner, if stage == "wire" { continue } - seedName := stage - if stage == "source" { - seedName = "source" - } - if err := syncSeedCorpus(filepath.Join(repository, "Compiler", "Fuzzing", "Corpus", seedName), corpus); err != nil { + if err := syncSeedCorpus(filepath.Join(repository, "Compiler", "Fuzzing", "Corpus", stage), corpus); err != nil { return fmt.Errorf("could not synchronize the versioned %s seed corpus: %w", stage, err) } } @@ -156,7 +151,7 @@ func runFuzzCampaign(repository string, currentHost host, runner commandRunner, macRuntime = runtime } smokeArguments, campaignArguments := fuzzBuildArguments(configuration, sanitizerConfiguration, macRuntime) - if err := runner.Run(repository, nil, bazel, smokeArguments...); err != nil { + if err := runner.Run(repository, nil, bazel, cachedBuild(smokeArguments)...); err != nil { return fmt.Errorf("could not build standalone fuzz smoke targets; preserved %q: %w", work, err) } if err := runner.Run(repository, selectedEnvironment, smoke, "-Write-Corpus", wireGenerated); err != nil { @@ -174,60 +169,29 @@ func runFuzzCampaign(repository string, currentHost host, runner commandRunner, } // Run smoke tests before changing Bazel's instrumentation configuration and // staging the campaign binaries into the same host output tree. - if err := runner.Run(repository, nil, bazel, campaignArguments...); err != nil { + if err := runner.Run(repository, nil, bazel, cachedBuild(campaignArguments)...); err != nil { return fmt.Errorf("could not build instrumented fuzz targets; preserved %q: %w", work, err) } - targets := []struct { - binary string - corpus string - maxLength string - rssLimit string - }{ - {"wire_fuzzer", "wire", "16384", "768"}, - {"lexer_fuzzer", "lexer", "65536", "1024"}, - {"parser_fuzzer", "parser", "65536", "1536"}, - {"source_llvm_fuzzer", "source", "65536", "4096"}, - } - var records []map[string]any - for _, target := range targets { - fuzzer := filepath.Join(repository, "bazel-bin", "Compiler", "Fuzzing", target.binary+currentHost.executable) - if strings.Contains(target.binary, "lexer") || strings.Contains(target.binary, "parser") || strings.Contains(target.binary, "source_llvm") { - if err := copyFile(frontendLibrary, filepath.Join(filepath.Dir(fuzzer), filepath.Base(frontendLibrary)), 0o755); err != nil { - return fmt.Errorf("could not stage frontend for %s; preserved %q: %w", target.binary, work, err) - } - } - campaignArtifacts := filepath.Join(artifacts, target.binary) - if err := os.MkdirAll(campaignArtifacts, 0o700); err != nil { - return fmt.Errorf("could not create %s artifact directory: %w", target.binary, err) - } - arguments := []string{ - filepath.Join(corpusRoot, target.corpus), - "-max_total_time=" + strconv.Itoa(duration), - "-max_len=" + target.maxLength, - "-timeout=30", - "-rss_limit_mb=" + target.rssLimit, - "-use_value_profile=1", - "-verbosity=0", - "-print_final_stats=1", - "-artifact_prefix=" + campaignArtifacts + string(os.PathSeparator), - } - output, runErr := runner.RunWithInput(repository, selectedEnvironment, "", fuzzer, arguments...) - if err := os.WriteFile(filepath.Join(artifacts, target.binary+".log"), []byte(output), 0o600); err != nil { - return fmt.Errorf("could not preserve campaign log: %w", err) - } - fmt.Print(output) - records = append(records, map[string]any{"target": target.binary, "seconds": duration, "rss_limit_mb": target.rssLimit, "sanitizer": sanitizerConfiguration, "native_coverage": true, "haskell_native_coverage": false, "success": runErr == nil}) - report, err := json.MarshalIndent(records, "", " ") - if err != nil { - return err - } - if err := os.WriteFile(filepath.Join(artifacts, "campaigns.json"), report, 0o600); err != nil { - return err - } - if runErr != nil { - return fmt.Errorf("%s campaign failed; corpus and crash artifacts preserved in %q: %w", target.binary, work, runErr) - } - fmt.Printf("%s campaign completed (%d seconds; RSS <= %s MiB).\n", target.binary, duration, target.rssLimit) + campaign := fuzzCampaign{ + repository: repository, + corpusRoot: corpusRoot, + artifacts: artifacts, + work: work, + report: "campaigns.json", + duration: duration, + environment: selectedEnvironment, + sanitizer: sanitizerConfiguration, + executable: currentHost.executable, + } + jobs, err := hostFuzzJobs(os.Getenv("VXS_FUZZ_JOBS")) + if err != nil { + return err + } + if err := campaign.runAll(runner, nativeFuzzTargets(), frontendLibrary, jobs); err != nil { + return err + } + if err := runHaskellFuzz(repository, corpusRoot, artifacts, duration, jobs, runner); err != nil { + return fmt.Errorf("Haskell feedback campaign failed; preserved %q: %w", work, err) } if os.Getenv("CI") == "true" { fmt.Printf("Campaign logs and coverage limits preserved in %s.\n", work) @@ -249,6 +213,96 @@ func runFuzzCampaign(repository string, currentHost host, runner commandRunner, return nil } +// fuzzCampaign carries the settings shared by every libFuzzer target of one +// helper invocation and accumulates their structured report. +type fuzzCampaign struct { + repository, corpusRoot, artifacts, work, report string + duration int + environment []string + sanitizer, executable string + // records holds one slot per target in inventory order; guard serializes + // report updates and console output of concurrently finishing targets. + records []map[string]any + guard sync.Mutex +} + +// runAll stages every frontend library first, because several targets share +// one output directory, and then runs the targets with bounded concurrency. +// Each target keeps its own corpus, artifact directory, time budget, RSS +// limit, watchdog and report checks exactly as in a sequential run. +func (campaign *fuzzCampaign) runAll(runner commandRunner, targets []fuzzTarget, frontendLibrary string, jobs int) error { + campaign.records = make([]map[string]any, len(targets)) + tasks := make([]fuzzTask, 0, len(targets)) + for index, target := range targets { + if target.frontend { + if err := copyFile(frontendLibrary, filepath.Join(filepath.Dir(campaign.program(target)), filepath.Base(frontendLibrary)), 0o755); err != nil { + return fmt.Errorf("could not stage frontend for %s; preserved %q: %w", target.binary, campaign.work, err) + } + } + tasks = append(tasks, fuzzTask{heavy: isHeavyFuzzTarget(target), run: func() error { return campaign.run(runner, index, target) }}) + } + fmt.Printf("Running %d fuzz targets with up to %d concurrent processes.\n", len(targets), jobs) + return runFuzzTasks(jobs, tasks) +} + +func (campaign *fuzzCampaign) program(target fuzzTarget) string { + packagePath, _, _ := strings.Cut(strings.TrimPrefix(target.label, "//"), ":") + return filepath.Join(campaign.repository, "bazel-bin", filepath.FromSlash(packagePath), target.binary+campaign.executable) +} + +// run executes one instrumented target against its persistent corpus. The +// report is rewritten after every target so a later failure still leaves the +// earlier measurements, and a process that exits zero without libFuzzer's final +// counters is a failure rather than an unexplained success. +func (campaign *fuzzCampaign) run(runner commandRunner, index int, target fuzzTarget) error { + fuzzer := campaign.program(target) + campaignArtifacts := filepath.Join(campaign.artifacts, target.binary) + if err := os.MkdirAll(campaignArtifacts, 0o700); err != nil { + return fmt.Errorf("could not create %s artifact directory: %w", target.binary, err) + } + arguments := []string{ + filepath.Join(campaign.corpusRoot, target.corpus), + "-max_total_time=" + strconv.Itoa(campaign.duration), + "-max_len=" + target.maxLength, + "-timeout=30", + "-rss_limit_mb=" + target.rssLimit, + "-use_value_profile=1", + "-verbosity=0", + "-print_final_stats=1", + "-artifact_prefix=" + campaignArtifacts + string(os.PathSeparator), + } + output, runErr := runner.RunWithInput(campaign.repository, campaign.environment, "", fuzzer, arguments...) + if err := os.WriteFile(filepath.Join(campaign.artifacts, target.binary+".log"), []byte(output), 0o600); err != nil { + return fmt.Errorf("could not preserve campaign log: %w", err) + } + statistics, statisticsErr := parseFuzzStatistics(output) + if runErr == nil && statisticsErr != nil { + runErr = statisticsErr + } + campaign.guard.Lock() + defer campaign.guard.Unlock() + fmt.Print(output) + campaign.records[index] = map[string]any{"target": target.binary, "seconds": campaign.duration, "rss_limit_mb": target.rssLimit, "sanitizer": campaign.sanitizer, "native_coverage": true, "haskell_native_coverage": false, "statistics": statistics, "success": runErr == nil} + finished := make([]map[string]any, 0, len(campaign.records)) + for _, record := range campaign.records { + if record != nil { + finished = append(finished, record) + } + } + report, err := json.MarshalIndent(finished, "", " ") + if err != nil { + return err + } + if err := os.WriteFile(filepath.Join(campaign.artifacts, campaign.report), report, 0o600); err != nil { + return err + } + if runErr != nil { + return fmt.Errorf("%s campaign failed; corpus and crash artifacts preserved in %q: %w", target.binary, campaign.work, runErr) + } + fmt.Printf("%s campaign completed (%d seconds; RSS <= %s MiB; %s).\n", target.binary, campaign.duration, target.rssLimit, campaign.sanitizer) + return nil +} + func fuzzDuration(stress bool, configured string) (int, error) { if configured == "" { if stress { diff --git a/helpers/internal/development/fuzz_haskell.go b/helpers/internal/development/fuzz_haskell.go new file mode 100644 index 00000000..fbba00db --- /dev/null +++ b/helpers/internal/development/fuzz_haskell.go @@ -0,0 +1,149 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "encoding/json" + "fmt" + "os" + "path/filepath" + "strconv" + "strings" + "sync" +) + +// The native C ABI wrapper cannot feed back GHC branches. This executable is +// rebuilt in an isolated HPC tree and retains inputs that hit new production +// ticks, rather than pretending native callback coverage represents Haskell. +func runHaskellFuzz(repository, corpusRoot, artifacts string, duration, jobs int, runner commandRunner) error { + directory := filepath.Join(repository, "Compiler") + options := []string{"--enable-coverage", "--disable-tests", "--builddir=dist-fuzz-coverage"} + build := append([]string{"build", "exe:frontend-fuzz"}, options...) + if err := runner.Run(directory, nil, "cabal", build...); err != nil { + return fmt.Errorf("HPC frontend build failed: %w", err) + } + arguments := append([]string{"list-bin", "exe:frontend-fuzz"}, options...) + binary, err := runner.OutputIn(directory, "cabal", arguments...) + if err != nil || !filepath.IsAbs(binary) || strings.ContainsAny(binary, "\r\n") { + return fmt.Errorf("could not resolve HPC frontend executable: %q (%v)", binary, err) + } + // Stages share only the read-only executable: each has its own corpus, + // tick file, report and artifact directory, so they may run together. + var console sync.Mutex + stages := []string{"lexer", "parser", "source"} + tasks := make([]fuzzTask, 0, len(stages)) + for _, stage := range stages { + tasks = append(tasks, fuzzTask{run: func() error { + corpus := filepath.Join(corpusRoot, "haskell-"+stage) + resultPath := filepath.Join(artifacts, "haskell-"+stage) + for _, path := range []string{corpus, resultPath} { + if err := os.MkdirAll(path, 0o700); err != nil { + return err + } + } + if err := syncSeedCorpus(filepath.Join(repository, "Compiler", "Fuzzing", "Corpus", stage), corpus); err != nil { + return err + } + // An HPC executable loads an existing tick file at startup and aborts + // when its module hashes belong to an earlier build. Give every stage a + // fresh file beside its report instead of the default one in the + // working directory, which would also leave output in the source tree. + tickFile := filepath.Join(resultPath, "frontend-fuzz.tix") + if err := os.Remove(tickFile); err != nil && !os.IsNotExist(err) { + return fmt.Errorf("could not reset the HPC tick file: %w", err) + } + // The engine writes its report to this file only after a completed + // campaign. Remove an earlier one so a crashed run cannot inherit it. + reportFile := filepath.Join(resultPath, "campaign.txt") + if err := os.Remove(reportFile); err != nil && !os.IsNotExist(err) { + return fmt.Errorf("could not reset the HPC campaign report: %w", err) + } + output, runErr := runner.RunWithInput(directory, []string{"HPCTIXFILE=" + tickFile}, "", binary, stage, strconv.Itoa(duration), corpus, resultPath, "12345") + console.Lock() + fmt.Print(output) + console.Unlock() + if err := os.WriteFile(filepath.Join(resultPath, "output.log"), []byte(output), 0o600); err != nil { + return err + } + stats, statsErr := readHaskellFuzzReport(reportFile, output, stage) + record := map[string]any{"stage": stage, "seconds": duration, "coverage_engine": "ghc-hpc", "asan_instrumented": false, "heap_limit_mib": 512, "statistics": stats, "success": runErr == nil && statsErr == nil} + report, err := json.MarshalIndent(record, "", " ") + if err != nil { + return err + } + if err := os.WriteFile(filepath.Join(resultPath, "campaign.json"), report, 0o600); err != nil { + return err + } + if runErr != nil { + return fmt.Errorf("HPC %s campaign failed; artifacts in %q: %w", stage, resultPath, runErr) + } + if statsErr != nil { + return statsErr + } + return nil + }}) + } + return runFuzzTasks(jobs, tasks) +} + +// readHaskellFuzzReport validates the report file the engine wrote for this +// run. Captured process output interleaves stdout with stderr, where the GHC +// runtime may print its own diagnostics after the report; those lines are kept +// in the log but are not statistics. The output must still announce the report, +// so a stale or hand-placed file is never accepted without a completed run. +func readHaskellFuzzReport(reportFile, output, stage string) (map[string]uint64, error) { + if !strings.Contains(output, "HPC_FUZZ_RESULT") { + return nil, fmt.Errorf("HPC %s campaign omitted its coverage report", stage) + } + report, err := os.ReadFile(reportFile) + if err != nil { + return nil, fmt.Errorf("HPC %s campaign did not write its report file: %w", stage, err) + } + return parseHaskellFuzzStatistics("HPC_FUZZ_RESULT\n"+string(report), stage) +} + +func parseHaskellFuzzStatistics(output, stage string) (map[string]uint64, error) { + _, report, found := strings.Cut(output, "HPC_FUZZ_RESULT\n") + if !found { // Windows process output may have CRLF newlines. + _, report, found = strings.Cut(output, "HPC_FUZZ_RESULT\r\n") + } + if !found { + return nil, fmt.Errorf("HPC %s campaign omitted its coverage report", stage) + } + values := make(map[string]uint64) + stageSeen := false + for _, line := range strings.Split(strings.ReplaceAll(report, "\r\n", "\n"), "\n") { + if line == "" { + continue + } + key, value, found := strings.Cut(line, "=") + if !found { + return nil, fmt.Errorf("malformed HPC report line: %q", line) + } + if key == "stage" { + if stageSeen || value != stage { + return nil, fmt.Errorf("unexpected or duplicate HPC stage: %q", value) + } + stageSeen = true + continue + } + if _, duplicate := values[key]; duplicate { + return nil, fmt.Errorf("duplicate HPC statistic: %s", key) + } + number, err := strconv.ParseUint(value, 10, 64) + if err != nil { + return nil, fmt.Errorf("invalid HPC statistic %s: %w", key, err) + } + values[key] = number + } + if !stageSeen || values["executed_units"] == 0 || values["covered_ticks"] == 0 || values["covered_ticks"] > values["available_ticks"] { + return nil, fmt.Errorf("HPC %s campaign did not prove production coverage", stage) + } + for _, key := range []string{"new_units_added", "seed"} { + if _, found := values[key]; !found { + return nil, fmt.Errorf("HPC report omitted %s", key) + } + } + return values, nil +} diff --git a/helpers/internal/development/fuzz_haskell_report_test.go b/helpers/internal/development/fuzz_haskell_report_test.go new file mode 100644 index 00000000..7e5c9219 --- /dev/null +++ b/helpers/internal/development/fuzz_haskell_report_test.go @@ -0,0 +1,47 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "os" + "path/filepath" + "strings" + "testing" +) + +func TestHaskellReportFileIgnoresRuntimeDiagnosticsButNotMissingEvidence(t *testing.T) { + report := "stage=lexer\nexecuted_units=3255\ncovered_ticks=839\navailable_ticks=36407\nnew_units_added=33\nseed=12345\n" + // The GHC runtime may print to stderr after the report; captured output + // interleaves both streams. + output := "HPC_FUZZ_RESULT\r\n" + strings.ReplaceAll(report, "\n", "\r\n") + + "\r\nonIOComplete: failed to grab table semaphore (res=2439, err=-1), dropping request 0x6\n" + file := filepath.Join(t.TempDir(), "campaign.txt") + if err := os.WriteFile(file, []byte(report), 0o600); err != nil { + t.Fatal(err) + } + values, err := readHaskellFuzzReport(file, output, "lexer") + if err != nil || values["covered_ticks"] != 839 || values["executed_units"] != 3255 { + t.Fatalf("completed campaign rejected: %v / %v", values, err) + } + // The same trailing diagnostics are still not statistics. + if _, err := parseHaskellFuzzStatistics(output, "lexer"); err == nil { + t.Fatal("runtime diagnostics were accepted as campaign statistics") + } + // A report file without a run that announced it is stale evidence. + if _, err := readHaskellFuzzReport(file, "process exited successfully", "lexer"); err == nil { + t.Fatal("report file accepted without a completed campaign") + } + if _, err := readHaskellFuzzReport(filepath.Join(t.TempDir(), "missing.txt"), output, "lexer"); err == nil { + t.Fatal("missing report file accepted") + } + if _, err := readHaskellFuzzReport(file, output, "parser"); err == nil { + t.Fatal("report for another stage accepted") + } + if err := os.WriteFile(file, []byte(strings.Replace(report, "covered_ticks=839", "covered_ticks=0", 1)), 0o600); err != nil { + t.Fatal(err) + } + if _, err := readHaskellFuzzReport(file, output, "lexer"); err == nil { + t.Fatal("report without production coverage accepted") + } +} diff --git a/helpers/internal/development/fuzz_haskell_test.go b/helpers/internal/development/fuzz_haskell_test.go new file mode 100644 index 00000000..4ed0fab6 --- /dev/null +++ b/helpers/internal/development/fuzz_haskell_test.go @@ -0,0 +1,70 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "strings" + "testing" +) + +func TestHaskellReportRequiresActualProductionCoverage(t *testing.T) { + valid := "HPC_FUZZ_RESULT\nstage=lexer\nexecuted_units=3378\ncovered_ticks=958\navailable_ticks=36407\nnew_units_added=38\nseed=12345\n" + for _, report := range []string{valid, strings.ReplaceAll(valid, "\n", "\r\n")} { + values, err := parseHaskellFuzzStatistics(report, "lexer") + if err != nil || values["covered_ticks"] != 958 { + t.Fatalf("valid coverage rejected: %v / %v", values, err) + } + } + for _, bad := range []string{ + "process exited successfully", + strings.Replace(valid, "stage=lexer", "stage=parser", 1), + strings.Replace(valid, "executed_units=3378", "executed_units=0", 1), + strings.Replace(valid, "covered_ticks=958", "covered_ticks=0", 1), + strings.Replace(valid, "available_ticks=36407", "available_ticks=10", 1), + strings.Replace(valid, "seed=12345\n", "", 1), + valid + "seed=54321\n", + valid + "stage=lexer\n", + strings.Replace(valid, "new_units_added=38", "new_units_added=-1", 1), + } { + if _, err := parseHaskellFuzzStatistics(bad, "lexer"); err == nil { + t.Fatalf("missing/malformed coverage report accepted: %q", bad) + } + } +} + +func TestNativeFuzzInventoryIsComponentOwnedAndBounded(t *testing.T) { + labels, corpora := map[string]bool{}, map[string]bool{} + for _, target := range nativeFuzzTargets() { + if labels[target.label] || corpora[target.corpus] { + t.Fatalf("duplicate target/corpus: %+v", target) + } + labels[target.label], corpora[target.corpus] = true, true + if !strings.HasPrefix(target.label, "//") || !strings.HasSuffix(target.label, ":"+target.binary) || target.maxLength == "" || target.rssLimit == "" { + t.Fatalf("unbounded or disconnected target: %+v", target) + } + } + for _, label := range []string{"//Compiler/Cli/Fuzzing:cli_fuzzer", "//Compiler/ProjectSystem/Bridge/Fuzzing:project_fuzzer", "//Interactive/Fuzzing:repl_fuzzer", "//Compiler/Runtime/AARC/Fuzzing:ownership_fuzzer"} { + if !labels[label] { + t.Fatalf("missing component-local target %s", label) + } + } +} + +func TestFuzzProcessWatchdogDoesNotLimitBuildTools(t *testing.T) { + for _, item := range []struct { + name string + args []string + seconds int + }{ + {"source_fuzz_smoke.exe", nil, 90}, + {"wire_fuzzer", []string{"-max_total_time=30"}, 120}, + {"frontend-fuzz.exe", []string{"parser", "900"}, 1200}, + {"cabal", []string{"build", "all"}, 0}, + {"bazel", []string{"build", "//Compiler:vxs"}, 0}, + } { + if got := fuzzProcessSeconds(item.name, item.args); got != item.seconds { + t.Fatalf("%s timeout = %d; want %d", item.name, got, item.seconds) + } + } +} diff --git a/helpers/internal/development/fuzz_parallel.go b/helpers/internal/development/fuzz_parallel.go new file mode 100644 index 00000000..f3c65eec --- /dev/null +++ b/helpers/internal/development/fuzz_parallel.go @@ -0,0 +1,108 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "errors" + "runtime" + "strconv" + "sync" +) + +// heavyRSSLimitMiB marks targets whose RSS limit is large enough that two of +// them running together could exhaust a developer machine. Their limits are +// not lowered; they simply never overlap each other. +const heavyRSSLimitMiB = 4096 + +// fuzzTask is one independent campaign process. Tasks share no corpus, +// artifact directory or report entry, so running them together changes only +// the wall-clock time, not the per-target budget or its checks. +type fuzzTask struct { + heavy bool + run func() error +} + +// fuzzJobs selects how many campaign processes may run at once. A campaign's +// evidence is the number of inputs a target executes inside its time budget, +// so concurrency is only free while every process keeps a physical core to +// itself. Measured on a two-core, four-thread host, two concurrent targets +// cut executed inputs by 40 to 90 percent. The default therefore grants one +// job per four logical processors, at most four, which is a single job on +// such a host and on four-vCPU CI runners. VXS_FUZZ_JOBS overrides it for a +// host with spare cores. +func fuzzJobs(configured string, logicalProcessors int) (int, error) { + if configured != "" { + jobs, err := strconv.Atoi(configured) + if err != nil || jobs < 1 || jobs > 64 { + return 0, errors.New("VXS_FUZZ_JOBS must be an integer in [1, 64]") + } + return jobs, nil + } + jobs := logicalProcessors / 4 + if jobs < 1 { + jobs = 1 + } + if jobs > 4 { + jobs = 4 + } + return jobs, nil +} + +func hostFuzzJobs(configured string) (int, error) { + return fuzzJobs(configured, runtime.NumCPU()) +} + +// runFuzzTasks executes every task with at most jobs running concurrently and +// at most one heavy task at a time. All started tasks finish so their logs and +// reports are complete; after the first failure no further task is started. +// The returned error is the failure of the earliest task in inventory order, +// which keeps the reported failure independent of scheduling. +func runFuzzTasks(jobs int, tasks []fuzzTask) error { + if jobs < 1 { + jobs = 1 + } + failures := make([]error, len(tasks)) + slots := make(chan struct{}, jobs) + var heavy sync.Mutex + var state sync.Mutex + failed := false + var group sync.WaitGroup + for index, task := range tasks { + slots <- struct{}{} + state.Lock() + stop := failed + state.Unlock() + if stop { + <-slots + break + } + group.Add(1) + go func() { + defer group.Done() + defer func() { <-slots }() + if task.heavy { + heavy.Lock() + defer heavy.Unlock() + } + if err := task.run(); err != nil { + state.Lock() + failures[index] = err + failed = true + state.Unlock() + } + }() + } + group.Wait() + for _, failure := range failures { + if failure != nil { + return failure + } + } + return nil +} + +func isHeavyFuzzTarget(target fuzzTarget) bool { + limit, err := strconv.Atoi(target.rssLimit) + return err != nil || limit >= heavyRSSLimitMiB +} diff --git a/helpers/internal/development/fuzz_parallel_test.go b/helpers/internal/development/fuzz_parallel_test.go new file mode 100644 index 00000000..04258cae --- /dev/null +++ b/helpers/internal/development/fuzz_parallel_test.go @@ -0,0 +1,129 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "errors" + "sync" + "sync/atomic" + "testing" + "time" +) + +func TestFuzzJobsDefaultKeepsAPhysicalCorePerTarget(t *testing.T) { + for _, example := range []struct{ processors, jobs int }{{1, 1}, {2, 1}, {4, 1}, {8, 2}, {16, 4}, {64, 4}} { + got, err := fuzzJobs("", example.processors) + if err != nil || got != example.jobs { + t.Fatalf("%d processors: %d jobs, %v; want %d", example.processors, got, err, example.jobs) + } + } + if got, err := fuzzJobs("1", 16); err != nil || got != 1 { + t.Fatalf("explicit sequential run rejected: %d, %v", got, err) + } + if got, err := fuzzJobs("6", 2); err != nil || got != 6 { + t.Fatalf("explicit job count rejected: %d, %v", got, err) + } + for _, bad := range []string{"0", "-1", "65", "two", "1.5"} { + if _, err := fuzzJobs(bad, 8); err == nil { + t.Fatalf("accepted invalid VXS_FUZZ_JOBS %q", bad) + } + } +} + +func TestFuzzTasksRespectJobAndHeavyLimits(t *testing.T) { + var running, heavyRunning, peak, heavyPeak atomic.Int32 + raise := func(current *atomic.Int32, highest *atomic.Int32) { + value := current.Add(1) + for { + seen := highest.Load() + if value <= seen || highest.CompareAndSwap(seen, value) { + return + } + } + } + var completed atomic.Int32 + tasks := make([]fuzzTask, 0, 12) + for index := 0; index < 12; index++ { + heavy := index%2 == 0 + tasks = append(tasks, fuzzTask{heavy: heavy, run: func() error { + raise(&running, &peak) + if heavy { + raise(&heavyRunning, &heavyPeak) + } + time.Sleep(20 * time.Millisecond) + if heavy { + heavyRunning.Add(-1) + } + running.Add(-1) + completed.Add(1) + return nil + }}) + } + if err := runFuzzTasks(3, tasks); err != nil { + t.Fatal(err) + } + if completed.Load() != 12 { + t.Fatalf("only %d of 12 tasks ran", completed.Load()) + } + if peak.Load() > 3 || peak.Load() < 2 { + t.Fatalf("concurrency peak %d is outside the requested bound of 3", peak.Load()) + } + if heavyPeak.Load() != 1 { + t.Fatalf("%d memory-heavy targets overlapped", heavyPeak.Load()) + } +} + +func TestFuzzTasksReportTheEarliestFailureAndStopStartingWork(t *testing.T) { + first, second := errors.New("first"), errors.New("second") + var started atomic.Int32 + var release sync.WaitGroup + release.Add(1) + tasks := []fuzzTask{ + {run: func() error { started.Add(1); release.Wait(); return first }}, + {run: func() error { started.Add(1); release.Done(); return second }}, + } + for index := 0; index < 20; index++ { + tasks = append(tasks, fuzzTask{run: func() error { started.Add(1); return nil }}) + } + err := runFuzzTasks(2, tasks) + // The later task fails first in time; the reported error is still the + // earliest one in inventory order. + if !errors.Is(err, first) { + t.Fatalf("reported %v instead of the earliest failure", err) + } + if started.Load() == int32(len(tasks)) { + t.Fatal("every task was started although the campaign had already failed") + } + // A single job is exactly the former sequential behavior. + order := []int{} + sequential := []fuzzTask{} + for index := 0; index < 5; index++ { + sequential = append(sequential, fuzzTask{heavy: index == 2, run: func() error { order = append(order, index); return nil }}) + } + if err := runFuzzTasks(1, sequential); err != nil || len(order) != 5 { + t.Fatalf("sequential run: %v, %v", order, err) + } + for index, value := range order { + if index != value { + t.Fatalf("one job did not preserve inventory order: %v", order) + } + } +} + +func TestOnlyLargeRSSTargetsAreHeavy(t *testing.T) { + heavy := map[string]bool{} + for _, target := range nativeFuzzTargets() { + heavy[target.binary] = isHeavyFuzzTarget(target) + } + for _, binary := range []string{"source_llvm_fuzzer", "differential_fuzzer", "repl_fuzzer"} { + if !heavy[binary] { + t.Fatalf("%s has a 4096 MiB limit but may overlap another heavy target", binary) + } + } + for _, binary := range []string{"wire_fuzzer", "lexer_fuzzer", "parser_fuzzer", "cli_fuzzer", "project_fuzzer", "ownership_fuzzer"} { + if heavy[binary] { + t.Fatalf("%s is needlessly serialized", binary) + } + } +} diff --git a/helpers/internal/development/fuzz_report.go b/helpers/internal/development/fuzz_report.go new file mode 100644 index 00000000..ff1eba50 --- /dev/null +++ b/helpers/internal/development/fuzz_report.go @@ -0,0 +1,61 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "fmt" + "strconv" + "strings" +) + +// Keep libFuzzer's counters beside the requested budget. Exit zero alone does +// not establish that a campaign executed any inputs or produced a final report. +type fuzzStatistics struct { + ExecutedUnits uint64 `json:"executed_units"` + AverageExecPerSec uint64 `json:"average_exec_per_second"` + NewUnitsAdded uint64 `json:"new_units_added"` + SlowestUnitSeconds uint64 `json:"slowest_unit_seconds"` + PeakRSSMiB uint64 `json:"peak_rss_mib"` +} + +func parseFuzzStatistics(output string) (*fuzzStatistics, error) { + statistics := &fuzzStatistics{} + fields := map[string]*uint64{ + "number_of_executed_units": &statistics.ExecutedUnits, + "average_exec_per_sec": &statistics.AverageExecPerSec, + "new_units_added": &statistics.NewUnitsAdded, + "slowest_unit_time_sec": &statistics.SlowestUnitSeconds, + "peak_rss_mb": &statistics.PeakRSSMiB, + } + seen := make(map[string]bool, len(fields)) + for _, line := range strings.Split(output, "\n") { + statistic, ok := strings.CutPrefix(strings.TrimSpace(line), "stat::") + if !ok { + continue + } + name, value, ok := strings.Cut(statistic, ":") + destination, recognized := fields[name] + if !recognized { + continue // Future libFuzzer counters do not change this contract. + } + if !ok || seen[name] { + return nil, fmt.Errorf("invalid or duplicate libFuzzer statistic %q", name) + } + parsed, err := strconv.ParseUint(strings.TrimSpace(value), 10, 64) + if err != nil { + return nil, fmt.Errorf("invalid libFuzzer statistic %q: %w", name, err) + } + *destination = parsed + seen[name] = true + } + for name := range fields { + if !seen[name] { + return nil, fmt.Errorf("libFuzzer did not report %q", name) + } + } + if statistics.ExecutedUnits == 0 || statistics.PeakRSSMiB == 0 { + return nil, fmt.Errorf("libFuzzer reported no executed input or no measured RSS") + } + return statistics, nil +} diff --git a/helpers/internal/development/fuzz_report_test.go b/helpers/internal/development/fuzz_report_test.go new file mode 100644 index 00000000..f87aa3ad --- /dev/null +++ b/helpers/internal/development/fuzz_report_test.go @@ -0,0 +1,46 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "strings" + "testing" +) + +const fuzzFinalStatistics = "INFO: seed corpus loaded\r\n" + + "stat::number_of_executed_units: 155\r\n" + + "stat::average_exec_per_sec: 5\r\n" + + "stat::new_units_added: 139\r\n" + + "stat::slowest_unit_time_sec: 0\r\n" + + "stat::peak_rss_mb: 196\r\n" + +func TestFuzzStatisticsPreserveExecutionEvidence(t *testing.T) { + statistics, err := parseFuzzStatistics(fuzzFinalStatistics + "stat::future_counter: 42\n") + if err != nil { + t.Fatal(err) + } + if statistics.ExecutedUnits != 155 || statistics.AverageExecPerSec != 5 || + statistics.NewUnitsAdded != 139 || statistics.SlowestUnitSeconds != 0 || statistics.PeakRSSMiB != 196 { + t.Fatalf("lost campaign evidence: %+v", statistics) + } +} + +func TestFuzzStatisticsRejectIncompleteOrContradictoryReports(t *testing.T) { + for name, report := range map[string]string{ + "missing": "INFO: clean exit without final statistics\n", + "truncated": strings.ReplaceAll(fuzzFinalStatistics, "stat::peak_rss_mb: 196\r\n", ""), + "duplicate": fuzzFinalStatistics + "stat::number_of_executed_units: 1\n", + "zero inputs": strings.ReplaceAll(fuzzFinalStatistics, "units: 155", "units: 0"), + "negative": strings.ReplaceAll(fuzzFinalStatistics, "units: 155", "units: -1"), + "overflow": strings.ReplaceAll(fuzzFinalStatistics, "units: 155", "units: 18446744073709551616"), + "fraction": strings.ReplaceAll(fuzzFinalStatistics, "units: 155", "units: 1.5"), + "no RSS": strings.ReplaceAll(fuzzFinalStatistics, "196", "0"), + } { + t.Run(name, func(t *testing.T) { + if _, err := parseFuzzStatistics(report); err == nil { + t.Fatal("accepted a report without reliable execution evidence") + } + }) + } +} diff --git a/helpers/internal/development/fuzz_targets.go b/helpers/internal/development/fuzz_targets.go new file mode 100644 index 00000000..45066c40 --- /dev/null +++ b/helpers/internal/development/fuzz_targets.go @@ -0,0 +1,43 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +// Labels, corpus identities and limits share one inventory so a component-local +// harness cannot silently disappear from the nightly run after a directory move. +type fuzzTarget struct { + label, binary, corpus, maxLength, rssLimit string + // frontend targets load the Haskell shared library beside the executable. + frontend bool + // threaded targets start their own worker threads. Only these gain evidence + // from a ThreadSanitizer campaign; the others execute one thread per input. + threaded bool +} + +func nativeFuzzTargets() []fuzzTarget { + return []fuzzTarget{ + {"//Compiler/Fuzzing:wire_fuzzer", "wire_fuzzer", "wire", "16384", "768", false, false}, + {"//Compiler/Fuzzing:lexer_fuzzer", "lexer_fuzzer", "lexer", "65536", "1024", true, false}, + {"//Compiler/Fuzzing:parser_fuzzer", "parser_fuzzer", "parser", "65536", "1536", true, false}, + {"//Compiler/Fuzzing:source_llvm_fuzzer", "source_llvm_fuzzer", "source", "65536", "4096", true, false}, + // The generator reads at most 33 bytes: 31 expression selectors at depth + // four, then one shape and one trip-count byte. Bound mutations close to + // that semantic input instead of evolving unused tails. + {"//Compiler/Fuzzing:differential_fuzzer", "differential_fuzzer", "differential", "64", "4096", true, false}, + {"//Compiler/Cli/Fuzzing:cli_fuzzer", "cli_fuzzer", "cli", "16384", "768", false, false}, + {"//Compiler/ProjectSystem/Bridge/Fuzzing:project_fuzzer", "project_fuzzer", "project", "65536", "768", false, false}, + {"//Interactive/Fuzzing:repl_fuzzer", "repl_fuzzer", "repl", "64", "4096", true, false}, + {"//Compiler/Runtime/AARC/Fuzzing:ownership_fuzzer", "ownership_fuzzer", "ownership", "64", "768", false, true}, + } +} + +// threadFuzzTargets selects the harnesses whose oracle depends on interleaving. +func threadFuzzTargets() []fuzzTarget { + var targets []fuzzTarget + for _, target := range nativeFuzzTargets() { + if target.threaded { + targets = append(targets, target) + } + } + return targets +} diff --git a/helpers/internal/development/fuzz_thread.go b/helpers/internal/development/fuzz_thread.go new file mode 100644 index 00000000..3e4eff92 --- /dev/null +++ b/helpers/internal/development/fuzz_thread.go @@ -0,0 +1,127 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "errors" + "fmt" + "os" + "path/filepath" +) + +// threadFuzzBuildArguments instruments the threaded harnesses for libFuzzer and +// ThreadSanitizer together. AddressSanitizer cannot share a binary with +// ThreadSanitizer, so this is a separate build and a separate campaign. +func threadFuzzBuildArguments(configuration, sanitizerConfiguration, macRuntime string) []string { + arguments := []string{"build", "--config=" + configuration, "--config=" + sanitizerConfiguration} + if macRuntime != "" { + arguments = append(arguments, "--linkopt="+macRuntime) + } + for _, target := range threadFuzzTargets() { + arguments = append(arguments, target.label) + } + return arguments +} + +// runThreadFuzzCampaign runs the interleaving-dependent harnesses under +// ThreadSanitizer. A host without a ThreadSanitizer runtime is an error, never +// a skipped success: the selection fails on Windows, and the runtime probe must +// start cleanly and report its intentional data race before any input runs. +func runThreadFuzzCampaign(repository string, currentHost host, runner commandRunner) error { + duration, err := fuzzDuration(false, os.Getenv("VXS_FUZZ_SECONDS")) + if err != nil { + return err + } + selected, err := selectSanitizer(currentHost, "thread") + if err != nil { + return err + } + configuration, err := fuzzConfiguration(currentHost) + if err != nil { + return err + } + targets := threadFuzzTargets() + if len(targets) == 0 { + return errors.New("no threaded fuzz target is registered") + } + bazel, err := findBazel(runner) + if err != nil { + return err + } + selected.environment, err = sanitizerEnvironment(currentHost, selected, runner) + if err != nil { + return err + } + if err := verifySanitizerRuntime(repository, currentHost, runner, selected); err != nil { + return err + } + temporaryRoot := os.Getenv("RUNNER_TEMP") + if temporaryRoot == "" { + temporaryRoot = os.TempDir() + } + work, err := os.MkdirTemp(temporaryRoot, "vxs-fuzz-") + if err != nil { + return fmt.Errorf("could not create fuzz work directory: %w", err) + } + persistentCorpus := os.Getenv("VXS_FUZZ_CORPUS") + corpusRoot := persistentCorpus + if corpusRoot == "" { + corpusRoot = filepath.Join(work, "corpus") + } + artifacts := filepath.Join(work, "artifacts") + if err := os.Mkdir(artifacts, 0o700); err != nil { + return fmt.Errorf("could not create fuzz artifact directory %q: %w", work, err) + } + for _, target := range targets { + corpus := filepath.Join(corpusRoot, target.corpus) + if err := os.MkdirAll(corpus, 0o700); err != nil { + return fmt.Errorf("could not prepare %s corpus; preserved %q: %w", target.corpus, work, err) + } + if err := syncSeedCorpus(filepath.Join(repository, "Compiler", "Fuzzing", "Corpus", target.corpus), corpus); err != nil { + return fmt.Errorf("could not synchronize the versioned %s seed corpus: %w", target.corpus, err) + } + } + macRuntime := "" + if currentHost.kind == hostMacOS { + macRuntime, err = macOSFuzzerRuntime(os.Getenv("LLVM_ROOT")) + if err != nil { + return fmt.Errorf("could not locate macOS libFuzzer runtime; preserved %q: %w", work, err) + } + } + if err := runner.Run(repository, nil, bazel, cachedBuild(threadFuzzBuildArguments(configuration, selected.config, macRuntime))...); err != nil { + return fmt.Errorf("could not build ThreadSanitizer fuzz targets; preserved %q: %w", work, err) + } + campaign := fuzzCampaign{ + repository: repository, + corpusRoot: corpusRoot, + artifacts: artifacts, + work: work, + report: "thread-campaigns.json", + duration: duration, + environment: selected.environment, + sanitizer: selected.config, + executable: currentHost.executable, + } + for _, target := range targets { + if target.frontend { + // The GHC runtime is not ThreadSanitizer-instrumented; a frontend + // target here would report races this campaign cannot attribute. + return fmt.Errorf("threaded fuzz target %s must not depend on the Haskell frontend", target.binary) + } + } + // Threaded targets already occupy several cores each and their findings + // depend on scheduling, so they run one at a time. + if err := campaign.runAll(runner, targets, "", 1); err != nil { + return err + } + if os.Getenv("CI") == "true" { + fmt.Printf("ThreadSanitizer campaign logs preserved in %s.\n", work) + return nil + } + if err := removeSuccessfulFuzzWork(temporaryRoot, work); err != nil { + return err + } + fmt.Println("All ThreadSanitizer fuzz targets completed without a reported failure.") + return nil +} diff --git a/helpers/internal/development/fuzz_thread_test.go b/helpers/internal/development/fuzz_thread_test.go new file mode 100644 index 00000000..3a88390b --- /dev/null +++ b/helpers/internal/development/fuzz_thread_test.go @@ -0,0 +1,44 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "strings" + "testing" +) + +func TestThreadFuzzPlanInstrumentsOnlyThreadedTargets(t *testing.T) { + targets := threadFuzzTargets() + if len(targets) != 1 || targets[0].binary != "ownership_fuzzer" || targets[0].frontend { + t.Fatalf("unexpected threaded targets: %+v", targets) + } + for _, example := range []struct{ fuzz, sanitizer, runtime string }{ + {"fuzz-linux", "tsan-linux", ""}, + {"fuzz-macos", "tsan-macos", "/llvm/libclang_rt.fuzzer_osx.a"}, + } { + plan := threadFuzzBuildArguments(example.fuzz, example.sanitizer, example.runtime) + joined := strings.Join(plan, " ") + for _, required := range []string{"--config=" + example.fuzz, "--config=" + example.sanitizer, "//Compiler/Runtime/AARC/Fuzzing:ownership_fuzzer"} { + if !strings.Contains(joined, required) { + t.Fatalf("thread plan lacks %s: %v", required, plan) + } + } + // ASan and TSan runtimes cannot coexist in one executable. + if strings.Contains(joined, "asan") || strings.Contains(joined, "_smoke") { + t.Fatalf("thread plan mixes incompatible instrumentation: %v", plan) + } + if (example.runtime != "") != strings.Contains(joined, "--linkopt=") { + t.Fatalf("libFuzzer runtime link option mismatch: %v", plan) + } + } +} + +func TestThreadFuzzIsRejectedWhereNoRuntimeExists(t *testing.T) { + // Windows has no Clang ThreadSanitizer runtime. The campaign must fail + // before building or executing anything instead of reporting success. + err := runThreadFuzzCampaign(t.TempDir(), host{kind: hostWindows, executable: ".exe"}, nil) + if err == nil || !strings.Contains(err.Error(), "does not support Windows ThreadSanitizer") { + t.Fatalf("unsupported host was not rejected explicitly: %v", err) + } +} diff --git a/helpers/internal/development/process.go b/helpers/internal/development/process.go index 3f737c39..a5820581 100644 --- a/helpers/internal/development/process.go +++ b/helpers/internal/development/process.go @@ -4,11 +4,16 @@ package development import ( + "context" + "fmt" "io" "os" "os/exec" + "path/filepath" "runtime" + "strconv" "strings" + "time" ) type commandRunner interface { @@ -25,24 +30,82 @@ type systemRunner struct { } func (runner systemRunner) Run(directory string, environment []string, name string, arguments ...string) error { - command := exec.Command(name, arguments...) + command, finish := watchedCommand(name, arguments) + defer finish(nil) command.Dir = directory command.Env = mergedEnvironment(environment) command.Stdin = os.Stdin command.Stdout = runner.stdout command.Stderr = runner.stderr - return command.Run() + return finish(command.Run()) } // RunWithInput drives programs with deterministic stdin while keeping captured // output available to the caller for smoke-test assertions. func (runner systemRunner) RunWithInput(directory string, environment []string, input string, name string, arguments ...string) (string, error) { - command := exec.Command(name, arguments...) + command, finish := watchedCommand(name, arguments) + defer finish(nil) command.Dir = directory command.Env = mergedEnvironment(environment) command.Stdin = strings.NewReader(input) output, err := command.CombinedOutput() - return string(output), err + return string(output), finish(err) +} + +// watchdogGrace bounds how long a cancelled program may hold its output pipes +// open after its process tree was terminated. +const watchdogGrace = 10 * time.Second + +// watchedCommand bounds fuzz and smoke programs. libFuzzer's -timeout covers one +// input, but a deterministic smoke program or an HPC campaign has no in-process +// watchdog, and a miscompiled generated loop never returns. The deadline +// terminates the whole process tree rather than only the direct child, and the +// returned finish function reports expiry as a watchdog failure instead of the +// host's generic kill status. Build tools are not bounded here; they keep their +// CI job timeout. +func watchedCommand(name string, arguments []string) (*exec.Cmd, func(error) error) { + return watchedCommandWithin(fuzzProcessSeconds(name, arguments), name, arguments) +} + +func watchedCommandWithin(seconds int, name string, arguments []string) (*exec.Cmd, func(error) error) { + if seconds == 0 { + return exec.Command(name, arguments...), func(err error) error { return err } + } + ctx, cancel := context.WithTimeout(context.Background(), time.Duration(seconds)*time.Second) + command := exec.CommandContext(ctx, name, arguments...) + prepareProcessTree(command) + command.Cancel = func() error { return terminateProcessTree(command) } + command.WaitDelay = watchdogGrace + return command, func(err error) error { + expired := ctx.Err() == context.DeadlineExceeded + cancel() + if err != nil && expired { + return fmt.Errorf("%s exceeded its %d-second process watchdog and its process tree was terminated: %w", filepath.Base(name), seconds, err) + } + return err + } +} + +func fuzzProcessSeconds(name string, arguments []string) int { + binary := strings.TrimSuffix(filepath.Base(name), ".exe") + if binary == "source_fuzz_smoke" || binary == "wire_fuzz_smoke" { + return 90 + } + if binary == "frontend-fuzz" && len(arguments) >= 2 { + if seconds, err := strconv.Atoi(arguments[1]); err == nil && seconds >= 1 && seconds <= 3600 { + return seconds + 300 // bounded corpus warmup and shutdown allowance + } + } + if strings.HasSuffix(binary, "_fuzzer") { + for _, argument := range arguments { + if value, ok := strings.CutPrefix(argument, "-max_total_time="); ok { + if seconds, err := strconv.Atoi(value); err == nil && seconds >= 1 && seconds <= 3600 { + return seconds + 90 + } + } + } + } + return 0 } // mergedEnvironment replaces inherited keys rather than appending duplicate diff --git a/helpers/internal/development/process_alive_unix_test.go b/helpers/internal/development/process_alive_unix_test.go new file mode 100644 index 00000000..596333a8 --- /dev/null +++ b/helpers/internal/development/process_alive_unix_test.go @@ -0,0 +1,35 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +//go:build !windows + +package development + +import ( + "os" + "strconv" + "strings" + "syscall" +) + +// processAlive reports whether a process with this identifier still executes. +// A killed orphan stays in the process table as a zombie until its new parent +// reaps it, and a container's first process may never do so; signal 0 still +// succeeds for such an entry. Where procfs is available, a zombie or dead +// state therefore counts as terminated. +func processAlive(pid int) bool { + if syscall.Kill(pid, 0) != nil { + return false + } + status, err := os.ReadFile("/proc/" + strconv.Itoa(pid) + "/stat") + if err != nil { + return true + } + // The command name is parenthesized and may itself contain spaces or + // parentheses; the state letter follows the last closing parenthesis. + _, rest, found := strings.Cut(string(status[strings.LastIndexByte(string(status), ')')+1:]), " ") + if !found || rest == "" { + return true + } + return rest[0] != 'Z' && rest[0] != 'X' +} diff --git a/helpers/internal/development/process_alive_windows_test.go b/helpers/internal/development/process_alive_windows_test.go new file mode 100644 index 00000000..d0c1fd64 --- /dev/null +++ b/helpers/internal/development/process_alive_windows_test.go @@ -0,0 +1,19 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +//go:build windows + +package development + +import ( + "os/exec" + "strconv" + "strings" +) + +// processAlive reports whether a process with this identifier still exists. +// os.FindProcess always succeeds on Windows, so query the process table. +func processAlive(pid int) bool { + output, err := exec.Command("tasklist", "/NH", "/FI", "PID eq "+strconv.Itoa(pid)).Output() + return err == nil && strings.Contains(string(output), " "+strconv.Itoa(pid)+" ") +} diff --git a/helpers/internal/development/process_test.go b/helpers/internal/development/process_test.go new file mode 100644 index 00000000..4b32bf09 --- /dev/null +++ b/helpers/internal/development/process_test.go @@ -0,0 +1,90 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package development + +import ( + "os" + "os/exec" + "path/filepath" + "strconv" + "strings" + "testing" + "time" +) + +const watchdogRole = "VXS_WATCHDOG_TEST_ROLE" + +// TestMain lets the test binary stand in for a hung fuzz program: the parent +// role starts a child that never exits and then blocks, like a smoke program +// stuck in a non-terminating generated loop with a helper process below it. +func TestMain(m *testing.M) { + switch os.Getenv(watchdogRole) { + case "parent": + child := exec.Command(os.Args[0]) + child.Env = append(os.Environ(), watchdogRole+"=child") + child.Stdout, child.Stderr = os.Stdout, os.Stderr + if err := child.Start(); err != nil { + os.Exit(3) + } + if err := os.WriteFile(os.Getenv("VXS_WATCHDOG_TEST_PID"), []byte(strconv.Itoa(child.Process.Pid)), 0o600); err != nil { + os.Exit(4) + } + time.Sleep(10 * time.Minute) + os.Exit(5) + case "child": + time.Sleep(10 * time.Minute) + os.Exit(6) + } + os.Exit(m.Run()) +} + +func TestWatchdogTerminatesTheWholeProcessTree(t *testing.T) { + pidFile := filepath.Join(t.TempDir(), "child.pid") + t.Setenv(watchdogRole, "parent") + t.Setenv("VXS_WATCHDOG_TEST_PID", pidFile) + command, finish := watchedCommandWithin(3, os.Args[0], nil) + started := time.Now() + output, runErr := command.CombinedOutput() + err := finish(runErr) + elapsed := time.Since(started) + if err == nil || !strings.Contains(err.Error(), "exceeded its 3-second process watchdog") { + t.Fatalf("hung program was not reported as a watchdog failure: %v\n%s", err, output) + } + // The deadline plus the pipe grace period bounds the wait; without tree + // termination the grandchild keeps the output pipe open for minutes. + if elapsed > 3*time.Second+watchdogGrace+5*time.Second { + t.Fatalf("watchdog returned only after %s", elapsed) + } + text, readErr := os.ReadFile(pidFile) + if readErr != nil { + t.Fatalf("parent never started its child: %v", readErr) + } + pid, convErr := strconv.Atoi(string(text)) + if convErr != nil { + t.Fatal(convErr) + } + deadline := time.Now().Add(10 * time.Second) + for processAlive(pid) { + if time.Now().After(deadline) { + t.Fatalf("descendant %d survived the watchdog", pid) + } + time.Sleep(100 * time.Millisecond) + } +} + +func TestWatchdogLeavesCompletedAndUnboundedCommandsAlone(t *testing.T) { + t.Setenv(watchdogRole, "") + for _, seconds := range []int{0, 60} { + command, finish := watchedCommandWithin(seconds, os.Args[0], []string{"-test.run=^$"}) + if err := finish(command.Run()); err != nil { + t.Fatalf("completed command reported a failure under a %d-second bound: %v", seconds, err) + } + } + // An ordinary failure inside the budget keeps its own error text. + command, finish := watchedCommandWithin(60, os.Args[0], []string{"-test.unknown-flag"}) + err := finish(command.Run()) + if err == nil || strings.Contains(err.Error(), "watchdog") { + t.Fatalf("ordinary failure was misreported: %v", err) + } +} diff --git a/helpers/internal/development/process_unix.go b/helpers/internal/development/process_unix.go new file mode 100644 index 00000000..ce2651f6 --- /dev/null +++ b/helpers/internal/development/process_unix.go @@ -0,0 +1,32 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +//go:build !windows + +package development + +import ( + "errors" + "os" + "os/exec" + "syscall" +) + +// prepareProcessTree places the child in its own process group so descendants +// that it spawns can be signalled together with it. +func prepareProcessTree(command *exec.Cmd) { + command.SysProcAttr = &syscall.SysProcAttr{Setpgid: true} +} + +// terminateProcessTree kills the child's whole process group. A group that has +// already exited is reported as os.ErrProcessDone, which exec treats as benign. +func terminateProcessTree(command *exec.Cmd) error { + if command.Process == nil { + return os.ErrProcessDone + } + err := syscall.Kill(-command.Process.Pid, syscall.SIGKILL) + if errors.Is(err, syscall.ESRCH) { + return os.ErrProcessDone + } + return err +} diff --git a/helpers/internal/development/process_windows.go b/helpers/internal/development/process_windows.go new file mode 100644 index 00000000..1510cef7 --- /dev/null +++ b/helpers/internal/development/process_windows.go @@ -0,0 +1,36 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +//go:build windows + +package development + +import ( + "os" + "os/exec" + "strconv" + "syscall" +) + +// prepareProcessTree detaches the child from the helper's console control +// group so terminating it cannot deliver a console event to the helper itself. +func prepareProcessTree(command *exec.Cmd) { + command.SysProcAttr = &syscall.SysProcAttr{CreationFlags: syscall.CREATE_NEW_PROCESS_GROUP} +} + +// terminateProcessTree ends the child and every descendant. Process.Kill alone +// leaves grandchildren running on Windows, where they keep the inherited output +// pipes open; taskkill /T walks the parent links instead. The direct child is +// killed as well in case taskkill itself is unavailable. +func terminateProcessTree(command *exec.Cmd) error { + if command.Process == nil { + return os.ErrProcessDone + } + tree := exec.Command("taskkill", "/T", "/F", "/PID", strconv.Itoa(command.Process.Pid)) + treeErr := tree.Run() + killErr := command.Process.Kill() + if treeErr == nil || killErr == nil { + return nil + } + return killErr +} diff --git a/helpers/internal/development/sanitizer.go b/helpers/internal/development/sanitizer.go index b35c1bc8..04bbe0a3 100644 --- a/helpers/internal/development/sanitizer.go +++ b/helpers/internal/development/sanitizer.go @@ -17,7 +17,7 @@ func verifySanitizerRuntime(repository string, currentHost host, runner commandR if err != nil { return err } - if err := runner.Run(repository, nil, bazel, "build", "--config="+selected.config, "//Compiler/Sanitizers:sanitizer_probe"); err != nil { + if err := runner.Run(repository, nil, bazel, cachedBuild([]string{"build", "--config=" + selected.config, "//Compiler/Sanitizers:sanitizer_probe"})...); err != nil { return fmt.Errorf("sanitizer probe build failed: %w", err) } probe := filepath.Join(repository, "bazel-bin", "Compiler", "Sanitizers", "sanitizer_probe"+currentHost.executable) diff --git a/justfile b/justfile index 896fd6b2..5c51c5e0 100644 --- a/justfile +++ b/justfile @@ -37,6 +37,10 @@ benchmark: fuzz: go run ./helpers/cmd/develop fuzz +# Threaded fuzz targets under ThreadSanitizer; macOS and native Linux only. +fuzz-thread: + go run ./helpers/cmd/develop fuzz-thread + fuzz-stress: go run ./helpers/cmd/develop fuzz-stress