From 16a214953eae5bd72a974fba946a45bf67f2179d Mon Sep 17 00:00:00 2001 From: Leitwolf11 Date: Fri, 2 Oct 2026 03:00:57 +0300 Subject: [PATCH] Check frontend and native CorePrep lowerings for parity in the source fuzz harness The Haskell frontend lowers optimized Core to CorePrep, and the native pipeline lowers the same Core again with its own adapter. Only the native result reaches Xpp and both lowerings are well formed, so a divergence is a miscompile no later verifier can see. Five such divergences were found by hand or by chance; this makes the comparison automatic. The testing-only frontend entry vxs_frontend_fuzz_compile now delivers the CorePrep wire of the same compilation as a second payload after Core. Production entries are unchanged and still deliver exactly one buffer, and the native receiver accepts CorePrep once, only after Core, and only on the testing route. CompareCorePrep brings both modules to a canonical form, reachable blocks in depth-first preorder with the true edge first and generated symbols renamed in first-use order while keeping their kind, and then compares source symbols, types, literals, operations, operand and instruction order and edge roles exactly. Its own test suite pins what the comparison may and may not ignore and joins the native suite list. Every accepted source in the source and differential fuzzers and in the deterministic smoke is checked, and the smoke gains four sources that must be accepted. Reintroducing the for-loop update defect fails the smoke in this check before any generated code runs. The fuzzing, Core IR and testing documents describe the check and its limits: it shows that the lowerings agree, not that either is correct. --- Compiler/Cli/Commands/Frontend.cpp | 54 +++- Compiler/Cli/Commands/Frontend.hpp | 22 +- Compiler/Fuzzing/BUILD.bazel | 14 + Compiler/Fuzzing/CorePrepParity.cpp | 301 ++++++++++++++++++ Compiler/Fuzzing/CorePrepParity.hpp | 34 ++ Compiler/Fuzzing/SourceFuzz.cpp | 48 ++- Compiler/Fuzzing/SourceFuzz.hpp | 9 +- Compiler/Fuzzing/SourceFuzzSmoke.cpp | 32 ++ Compiler/Fuzzing/Tests/BUILD.bazel | 14 + .../Fuzzing/Tests/CorePrepParityTests.cpp | 295 +++++++++++++++++ .../Driver/src/Visual/XSharp/Driver/FFI.hs | 32 +- Compiler/Headers/Visual/XSharp/Frontend.h | 19 +- Documents/CORE-IR.md | 5 +- Documents/FUZZING.md | 26 ++ Documents/TESTING.md | 2 +- helpers/internal/development/build.go | 1 + 16 files changed, 895 insertions(+), 13 deletions(-) create mode 100644 Compiler/Fuzzing/CorePrepParity.cpp create mode 100644 Compiler/Fuzzing/CorePrepParity.hpp create mode 100644 Compiler/Fuzzing/Tests/BUILD.bazel create mode 100644 Compiler/Fuzzing/Tests/CorePrepParityTests.cpp diff --git a/Compiler/Cli/Commands/Frontend.cpp b/Compiler/Cli/Commands/Frontend.cpp index 81e8b29f..d60b4b14 100644 --- a/Compiler/Cli/Commands/Frontend.cpp +++ b/Compiler/Cli/Commands/Frontend.cpp @@ -59,6 +59,11 @@ namespace Visual::XSharp::Cli::Frontend std::optional kind; std::vector bytes; std::string error; + // Only the testing route sets this. It lets one CorePrep payload + // follow a Core payload; every other route keeps the + // single-payload contract. + bool acceptCorePrep{}; + std::optional> corePrep; }; auto @@ -313,10 +318,26 @@ namespace Visual::XSharp::Cli::Frontend const std::uint8_t *bytes, std::size_t size) noexcept -> std::int32_t { - if (context == nullptr || rawKind > 3U + if (context == nullptr || rawKind > 4U || (size != 0U && bytes == nullptr)) return 1; auto &captured = *static_cast(context); + if (static_cast(rawKind) == OutputKind::CorePrepWire) + { + // CorePrep is accepted once, only after Core, and only on + // the route that asked for it. + if (!captured.acceptCorePrep + || captured.kind != OutputKind::CoreWire + || captured.corePrep.has_value() + || size > kMaximumCoreBytes) + { + captured.error = "frontend emitted an unexpected CorePrep " + "payload"; + return 1; + } + captured.corePrep.emplace(bytes, bytes + size); + return 0; + } if (captured.kind.has_value()) { captured.error @@ -450,8 +471,39 @@ namespace Visual::XSharp::Cli::Frontend {}, frontend.Error() }; CapturedOutput output; + output.acceptCorePrep = true; const auto status = frontend.FuzzCompile(source.data(), source.size(), output); return MakeResult(status, std::move(output)); } + + auto + FuzzCompileStages(std::span source) -> StageResult + { + auto &frontend = GetFrontend(); + if (!frontend.IsReady()) + return { { Status::InternalError, + OutputKind::ErrorText, + {}, + frontend.Error() }, + {} }; + CapturedOutput output; + output.acceptCorePrep = true; + const auto status + = frontend.FuzzCompile(source.data(), source.size(), output); + auto corePrep = std::move(output.corePrep); + StageResult result{ MakeResult(status, std::move(output)), {} }; + if (result.core.succeeded() && result.core.kind == OutputKind::CoreWire) + { + if (!corePrep) + { + result.core.status = Status::InternalError; + result.core.error + = "frontend delivered Core without its CorePrep lowering"; + return result; + } + result.corePrep = std::move(*corePrep); + } + return result; + } } // namespace Visual::XSharp::Cli::Frontend diff --git a/Compiler/Cli/Commands/Frontend.hpp b/Compiler/Cli/Commands/Frontend.hpp index ee0ae730..7271f4ef 100644 --- a/Compiler/Cli/Commands/Frontend.hpp +++ b/Compiler/Cli/Commands/Frontend.hpp @@ -19,7 +19,8 @@ namespace Visual::XSharp::Cli::Frontend CoreWire = 0, ///< Versioned, verified Core wire bytes. ProjectSourceList = 1, ///< NUL-delimited UTF-8 source paths. DiagnosticWire = 2, ///< Structured source diagnostics. - ErrorText = 3 ///< Human-readable UTF-8 failure text. + ErrorText = 3, ///< Human-readable UTF-8 failure text. + CorePrepWire = 4 ///< Frontend CorePrep lowering; testing only. }; /// @brief Stable operation outcomes returned by the C ABI. @@ -79,4 +80,23 @@ namespace Visual::XSharp::Cli::Frontend /// failures. [[nodiscard]] auto FuzzCompile(std::span source) -> Result; + + /// @brief Core and the frontend's own CorePrep from one compilation. + struct StageResult final + { + Result core; ///< Status and Core wire, as returned by FuzzCompile. + /// Frontend-lowered CorePrep wire; empty unless core succeeded. + std::vector corePrep; + }; + + /// @brief Compile source bytes and keep both frontend artifacts. + /// + /// The native pipeline lowers Core to CorePrep with its own adapter. This + /// testing route additionally returns the Haskell lowering of the same + /// Core so the two can be compared. + /// @param source Candidate source bytes; malformed input is permitted. + /// @return Owned Core and CorePrep buffers; a successful status without + /// CorePrep is reported as an internal error. + [[nodiscard]] auto + FuzzCompileStages(std::span source) -> StageResult; } // namespace Visual::XSharp::Cli::Frontend diff --git a/Compiler/Fuzzing/BUILD.bazel b/Compiler/Fuzzing/BUILD.bazel index 0d366e82..14278ddc 100644 --- a/Compiler/Fuzzing/BUILD.bazel +++ b/Compiler/Fuzzing/BUILD.bazel @@ -33,14 +33,28 @@ cc_binary( deps = [":wire_fuzz_harness"], ) +# Structural comparison of two CorePrep lowerings. It has no frontend or LLVM +# dependency so its own contract can be tested without either. +cc_library( + name = "coreprep_parity", + srcs = ["CorePrepParity.cpp"], + hdrs = ["CorePrepParity.hpp"], + visibility = ["//Compiler/Fuzzing/Tests:__pkg__"], + deps = ["//Compiler/Headers/Visual/XSharp/Core:core"], +) + cc_library( name = "source_fuzz_harness", srcs = ["SourceFuzz.cpp"], hdrs = ["SourceFuzz.hpp"], deps = [ + ":coreprep_parity", "//Compiler/Backend/LLVM:llvm_backend", "//Compiler/Cli/Commands:frontend", + "//Compiler/Core:core", "//Compiler/Driver:pipeline", + "//Compiler/Headers/Visual/XSharp/Core:core", + "//Compiler/Headers/Visual/XSharp/Core:coreprep_wire", ], ) diff --git a/Compiler/Fuzzing/CorePrepParity.cpp b/Compiler/Fuzzing/CorePrepParity.cpp new file mode 100644 index 00000000..44b907c8 --- /dev/null +++ b/Compiler/Fuzzing/CorePrepParity.cpp @@ -0,0 +1,301 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include + +#include "CorePrepParity.hpp" + +namespace Visual::XSharp::Fuzzing +{ + namespace + { + namespace Prepared = ::visual_xsharp::core; + + // Canonical identities start far above any real symbol so a renamed + // generated symbol can never equal a source symbol. + constexpr Prepared::SymbolId kCanonicalBase = Prepared::SymbolId{ 1 } + << 48U; + + /// Generated symbols are spelled with a leading `$`, which cannot + /// occur in a source identifier. + [[nodiscard]] auto + IsGenerated(const Prepared::SymbolName &symbol) -> bool + { + return !symbol.spelling.empty() && symbol.spelling.front() == U'$'; + } + + /// `$coreprep17` and `$coreprep4` are the same kind of temporary; + /// the digits only repeat the identity. + [[nodiscard]] auto + KindOf(const std::u32string &spelling) -> std::u32string + { + auto end = spelling.size(); + while (end > 0U && spelling[end - 1U] >= U'0' + && spelling[end - 1U] <= U'9') + --end; + return spelling.substr(0U, end); + } + + /** + * @brief Renames generated symbols in first-use order. + * + * Both lowerings may number temporaries differently, and one module + * shares one numbering because a lifted closure is named in its + * parent and defined later. Source symbols are never renamed. + */ + class Renamer final + { + public: + void + Apply(Prepared::SymbolName &symbol) + { + if (!IsGenerated(symbol)) + return; + const auto [found, inserted] + = names_.try_emplace(symbol.id, Prepared::SymbolName{}); + if (inserted) + found->second = { kCanonicalBase + names_.size(), + KindOf(symbol.spelling) }; + symbol = found->second; + } + + void + Apply(Prepared::Atom &atom) + { + if (atom.kind == Prepared::Atom::Kind::Variable) + Apply(atom.symbol); + } + + private: + std::unordered_map names_; + }; + + [[nodiscard]] auto + Successors(const Prepared::Block &block) + -> std::vector + { + switch (block.terminator.kind) + { + case Prepared::Terminator::Kind::Jump: + return { block.terminator.true_target }; + case Prepared::Terminator::Kind::Branch: + return { block.terminator.true_target, + block.terminator.false_target }; + case Prepared::Terminator::Kind::Return: + case Prepared::Terminator::Kind::Unreachable: + break; + } + return {}; + } + + /** + * @brief Reorders reachable blocks in depth-first preorder. + * + * Block identities are allocation artifacts. Following the true + * edge before the false edge from the entry gives both lowerings + * the same order exactly when their control-flow graphs are + * isomorphic with matching edge roles. Blocks that cannot be + * reached from the entry, such as the join after two returning + * branches, never execute and are left out. + */ + void + CanonicalizeBlocks(Prepared::Function &function) + { + std::unordered_map position; + for (std::size_t index = 0U; index < function.blocks.size(); + ++index) + position.emplace(function.blocks[index].id, index); + + std::unordered_map renumbered; + std::vector order; + std::vector pending{ function.entry }; + while (!pending.empty()) + { + const auto id = pending.back(); + pending.pop_back(); + const auto found = position.find(id); + if (found == position.end() || renumbered.contains(id)) + continue; + renumbered.emplace( + id, + static_cast(order.size())); + order.push_back(found->second); + auto successors = Successors(function.blocks[found->second]); + // The stack reverses order, so push the false edge first. + std::ranges::reverse(successors); + pending.insert(pending.end(), + successors.begin(), + successors.end()); + } + + const auto target = [&renumbered](Prepared::BlockId id) { + const auto found = renumbered.find(id); + // A dangling target is a verifier matter; keep it distinct + // from every canonical block. + return found == renumbered.end() ? ~Prepared::BlockId{} + : found->second; + }; + std::vector blocks; + blocks.reserve(order.size()); + for (const auto index : order) + { + auto block = std::move(function.blocks[index]); + block.id = static_cast(blocks.size()); + switch (block.terminator.kind) + { + case Prepared::Terminator::Kind::Branch: + block.terminator.false_target + = target(block.terminator.false_target); + [[fallthrough]]; + case Prepared::Terminator::Kind::Jump: + block.terminator.true_target + = target(block.terminator.true_target); + break; + case Prepared::Terminator::Kind::Return: + case Prepared::Terminator::Kind::Unreachable: + break; + } + blocks.push_back(std::move(block)); + } + function.blocks = std::move(blocks); + function.entry = 0U; + } + + [[nodiscard]] auto + Canonicalize(Prepared::CorePrepModule module) + -> Prepared::CorePrepModule + { + Renamer renamer; + for (auto &function : module.functions) + { + CanonicalizeBlocks(function); + renamer.Apply(function.symbol); + for (auto ¶meter : function.parameters) + renamer.Apply(parameter.symbol); + for (auto &block : function.blocks) + { + for (auto &instruction : block.instructions) + { + // Operands are read before the destination is + // written, so name them first. + for (auto &operand : instruction.operands) + renamer.Apply(operand); + for (auto &capture : instruction.captures) + { + renamer.Apply(capture.value); + renamer.Apply(capture.symbol); + } + renamer.Apply(instruction.closure_function); + renamer.Apply(instruction.destination); + } + renamer.Apply(block.terminator.value); + } + } + return module; + } + + [[nodiscard]] auto + Narrow(const std::u32string &text) -> std::string + { + std::string result; + for (const auto point : text) + result.push_back(point < 0x80U ? static_cast(point) + : '?'); + return result; + } + + [[nodiscard]] auto + Describe(const Prepared::Instruction &instruction) -> std::string + { + return "kind " + + std::to_string(static_cast(instruction.kind)) + + ", operation " + + std::to_string( + static_cast(instruction.operation)) + + ", destination " + Narrow(instruction.destination.spelling) + + ", " + std::to_string(instruction.operands.size()) + + " operand(s)"; + } + + [[nodiscard]] auto + Describe(const Prepared::Terminator &terminator) -> std::string + { + return "terminator kind " + + std::to_string(static_cast(terminator.kind)) + + " -> " + std::to_string(terminator.true_target) + "/" + + std::to_string(terminator.false_target); + } + + [[nodiscard]] auto + FirstDifference(const Prepared::Function &frontend, + const Prepared::Function &native) -> std::string + { + if (frontend.symbol != native.symbol + || frontend.parameters != native.parameters + || frontend.return_type != native.return_type + || frontend.sourceFile != native.sourceFile) + return "signature or source owner differs"; + if (frontend.blocks.size() != native.blocks.size()) + return "reachable block count " + + std::to_string(frontend.blocks.size()) + + " (frontend) versus " + + std::to_string(native.blocks.size()) + " (native)"; + for (std::size_t block = 0U; block < frontend.blocks.size(); + ++block) + { + const auto &left = frontend.blocks[block]; + const auto &right = native.blocks[block]; + const auto where = "canonical block " + std::to_string(block); + const auto shared = std::min(left.instructions.size(), + right.instructions.size()); + for (std::size_t index = 0U; index < shared; ++index) + if (left.instructions[index] != right.instructions[index]) + return where + ", instruction " + std::to_string(index) + + ": frontend {" + + Describe(left.instructions[index]) + + "} versus native {" + + Describe(right.instructions[index]) + "}"; + if (left.instructions.size() != right.instructions.size()) + return where + ": " + + std::to_string(left.instructions.size()) + + " instruction(s) (frontend) versus " + + std::to_string(right.instructions.size()) + + " (native)"; + if (left.terminator != right.terminator) + return where + ": frontend {" + Describe(left.terminator) + + "} versus native {" + Describe(right.terminator) + + "}"; + } + return "functions differ outside their blocks"; + } + } // namespace + + auto + CompareCorePrep(const Prepared::CorePrepModule &frontend, + const Prepared::CorePrepModule &native) + -> std::optional + { + const auto left = Canonicalize(frontend); + const auto right = Canonicalize(native); + if (left == right) + return std::nullopt; + if (left.name != right.name || left.sourceFiles != right.sourceFiles) + return "module name or source catalog differs"; + if (left.functions.size() != right.functions.size()) + return "function count " + std::to_string(left.functions.size()) + + " (frontend) versus " + + std::to_string(right.functions.size()) + " (native)"; + for (std::size_t index = 0U; index < left.functions.size(); ++index) + if (left.functions[index] != right.functions[index]) + return "function " + std::to_string(index) + " (" + + Narrow(left.functions[index].symbol.spelling) + "): " + + FirstDifference(left.functions[index], + right.functions[index]); + return "modules differ"; + } +} // namespace Visual::XSharp::Fuzzing diff --git a/Compiler/Fuzzing/CorePrepParity.hpp b/Compiler/Fuzzing/CorePrepParity.hpp new file mode 100644 index 00000000..9738cac5 --- /dev/null +++ b/Compiler/Fuzzing/CorePrepParity.hpp @@ -0,0 +1,34 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +#pragma once + +#include +#include + +#include "Visual/XSharp/Core/CorePrep.hpp" + +namespace Visual::XSharp::Fuzzing +{ + /** + * @brief Compare two CorePrep lowerings of the same Core module. + * + * The Haskell frontend and the native Core-to-CorePrep adapter must + * build the same program: the same reachable control-flow graph with the + * same instructions in every block. Block identities and the identities + * of generated symbols are allocation details, so both modules are first + * brought to a canonical form: reachable blocks in depth-first preorder + * from the entry, true edge before false edge, and `$`-prefixed generated + * symbols renamed in first-use order while keeping their kind. Source + * symbols, types, literals, operations, operand order, instruction order + * and edge roles are compared exactly. Blocks unreachable from the entry + * are ignored. + * + * @param frontend CorePrep produced by the Haskell frontend. + * @param native CorePrep produced by the native adapter. + * @return std::nullopt when equivalent, otherwise the first difference. + */ + [[nodiscard]] auto + CompareCorePrep(const ::visual_xsharp::core::CorePrepModule &frontend, + const ::visual_xsharp::core::CorePrepModule &native) + -> std::optional; +} // namespace Visual::XSharp::Fuzzing diff --git a/Compiler/Fuzzing/SourceFuzz.cpp b/Compiler/Fuzzing/SourceFuzz.cpp index 009250b8..ed862872 100644 --- a/Compiler/Fuzzing/SourceFuzz.cpp +++ b/Compiler/Fuzzing/SourceFuzz.cpp @@ -11,6 +11,7 @@ #include #include "Compiler/Cli/Commands/Frontend.hpp" +#include "CorePrepParity.hpp" #include "SourceFuzz.hpp" #include "Visual/XSharp/Backend/LLVM.hpp" #include "Visual/XSharp/Pipeline.hpp" @@ -167,6 +168,17 @@ namespace Visual::XSharp::Fuzzing "}\n"; } + /** + * @brief Compile once and require both CorePrep lowerings to agree. + * + * The frontend lowers its optimized Core to CorePrep, and the native + * pipeline lowers the same Core again with its own adapter. Only + * the native result reaches Xpp, so a divergence is a miscompile + * that no later verifier can see: both lowerings are well formed. + * This check decodes the Core and CorePrep buffers of one + * compilation, runs the native adapter, and fails on the first + * structural difference. + */ [[nodiscard]] auto CompileSource(std::span source) -> Frontend::Result { @@ -175,7 +187,34 @@ namespace Visual::XSharp::Fuzzing Frontend::OutputKind::ErrorText, {}, "source fuzz input exceeds 64 KiB" }; - return Frontend::FuzzCompile(source); + auto stages = Frontend::FuzzCompileStages(source); + if (!stages.core.succeeded() + || stages.core.kind != Frontend::OutputKind::CoreWire) + return std::move(stages.core); + + const auto core = Core::Wire::Decode(stages.core.bytes); + if (!core) + llvm::report_fatal_error(llvm::Twine( + "native Core reader rejected frontend Core wire")); + // Unverified Core is rejected by the pipeline with its own + // report; lowering it here would compare undefined shapes. + if (!Core::Verify(*core.module).empty()) + return std::move(stages.core); + const auto frontendCorePrep + = ::visual_xsharp::core::wire::decode(stages.corePrep); + if (!frontendCorePrep) + llvm::report_fatal_error(llvm::Twine( + "native CorePrep reader rejected frontend CorePrep wire")); + const auto difference + = CompareCorePrep(*frontendCorePrep.module, + Core::CorePrep::Prepare(*core.module)); + if (difference) + llvm::report_fatal_error(llvm::Twine( + "frontend and native CorePrep lowerings differ: " + + *difference + "; source:\n" + + std::string(reinterpret_cast(source.data()), + source.size()))); + return std::move(stages.core); } [[nodiscard]] auto @@ -352,6 +391,13 @@ namespace Visual::XSharp::Fuzzing (void)ConsumeVerifiedCore(compiled, "arbitrary source fuzz input"); } + void + ExerciseAcceptedSource(std::span input) + { + const auto compiled = CompileSource(input); + (void)ConsumeVerifiedCore(compiled, "source that must be accepted"); + } + void ExerciseDifferentialOracle(std::span input) { diff --git a/Compiler/Fuzzing/SourceFuzz.hpp b/Compiler/Fuzzing/SourceFuzz.hpp index ced12136..d86456f2 100644 --- a/Compiler/Fuzzing/SourceFuzz.hpp +++ b/Compiler/Fuzzing/SourceFuzz.hpp @@ -9,7 +9,9 @@ namespace Visual::XSharp::Fuzzing { // These routes deliberately share production Haskell entry points. The // syntax stages stop after the lexer or parser; source compilation goes - // through verified Core, Xpp, Xmm and LLVM lowering. + // through verified Core, Xpp, Xmm and LLVM lowering. Every accepted + // source is also lowered to CorePrep by both the frontend and the native + // adapter, and the two results must be structurally equal. void ExerciseLexer(std::span input); void @@ -18,4 +20,9 @@ namespace Visual::XSharp::Fuzzing ExerciseSourceToLlvm(std::span input); void ExerciseDifferentialOracle(std::span input); + /// Like ExerciseSourceToLlvm, but the source is known to be valid: a + /// frontend rejection is a failure. Deterministic checks use this so a + /// program that silently stopped compiling cannot pass as "rejected". + void + ExerciseAcceptedSource(std::span input); } // namespace Visual::XSharp::Fuzzing diff --git a/Compiler/Fuzzing/SourceFuzzSmoke.cpp b/Compiler/Fuzzing/SourceFuzzSmoke.cpp index 52ae85bb..79d860e7 100644 --- a/Compiler/Fuzzing/SourceFuzzSmoke.cpp +++ b/Compiler/Fuzzing/SourceFuzzSmoke.cpp @@ -24,6 +24,38 @@ main() const std::span emptySource; Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(emptySource); Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(source); + // Control-flow shapes whose frontend and native CorePrep lowerings must + // agree. They combine forms the generated programs below keep separate: + // loops inside loops, short-circuit operators as loop conditions, and + // several functions sharing one module-wide symbol numbering. + constexpr std::array accepted{ + "namespace Parity; class Program { public static int Evaluate() { " + "int total = 0; for (int outer = 0; outer < 4; outer++) { " + "if (outer == 2) { continue; } int inner = 0; " + "while (inner < 3) { if (inner == 1) { inner++; continue; } " + "total = total + outer + inner; inner++; } } return total; } }", + "namespace Parity; class Program { public static int Evaluate() { " + "int index = 0; int total = 0; " + "while (index < 9 && (index < 4 || total < 20)) { " + "total = total + index; index++; } " + "do { total++; } while (total < 40 && index \\= 0); " + "return total; } }", + "namespace Parity; class Program { " + "public static bool Down(_ int n) { return n == 0 || Down(n - 1); } " + "public static int Pick(_ int n) { if (Down(n) && n < 6) { " + "return n; } return 0 - 1; } " + "public static int Evaluate() { return Pick(3) + Pick(8); } }", + "namespace Parity; class First { public static long Evaluate() { " + "long result = 8; if (result > 4) { result = result - 2; } " + "return result; } } class Second { public static long Other() { " + "long value = 3; for (int step = 0; step < 2; step++) { " + "value = value * 2; } return value; } }", + }; + for (const auto text : accepted) + Visual::XSharp::Fuzzing::ExerciseAcceptedSource( + std::span( + reinterpret_cast(text.data()), + text.size())); llvm::errs() << "Differential smoke: mixed seed\n"; Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(expressionSeed); llvm::errs() << "Differential smoke: empty seed\n"; diff --git a/Compiler/Fuzzing/Tests/BUILD.bazel b/Compiler/Fuzzing/Tests/BUILD.bazel new file mode 100644 index 00000000..64fcbd12 --- /dev/null +++ b/Compiler/Fuzzing/Tests/BUILD.bazel @@ -0,0 +1,14 @@ +load("@rules_cc//cc:cc_binary.bzl", "cc_binary") + +package(default_visibility = ["//visibility:private"]) + +cc_binary( + name = "coreprep_parity_tests", + srcs = ["CorePrepParityTests.cpp"], + deps = [ + "//Compiler/Fuzzing:coreprep_parity", + "//Compiler/Headers/Visual/XSharp/Core:core", + "@catch3//:catch2_main", + "@catch3//src/Progmasoft:catch3", + ], +) diff --git a/Compiler/Fuzzing/Tests/CorePrepParityTests.cpp b/Compiler/Fuzzing/Tests/CorePrepParityTests.cpp new file mode 100644 index 00000000..f6fbaeb5 --- /dev/null +++ b/Compiler/Fuzzing/Tests/CorePrepParityTests.cpp @@ -0,0 +1,295 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include + +#include "Compiler/Fuzzing/CorePrepParity.hpp" +#include "Visual/XSharp/Core/CorePrep.hpp" + +// The parity oracle is only as strong as its notion of "same program". These +// tests pin what it must ignore (block and temporary numbering, unreachable +// blocks) and, more importantly, what it must never ignore. + +namespace +{ + namespace Prepared = visual_xsharp::core; + using Visual::XSharp::Fuzzing::CompareCorePrep; + + [[nodiscard]] auto + Symbol(std::uint64_t id, std::u32string spelling) -> Prepared::SymbolName + { + return { id, std::move(spelling) }; + } + + [[nodiscard]] auto + Read(std::uint64_t id, std::u32string spelling, Prepared::Type type) + -> Prepared::Atom + { + return Prepared::Atom::variable(Symbol(id, std::move(spelling)), + std::move(type)); + } + + [[nodiscard]] auto + Number(std::int64_t value) -> Prepared::Atom + { + return Prepared::Atom::constant(value, Prepared::Type::int64()); + } + + [[nodiscard]] auto + Bind(Prepared::SymbolName destination, + Prepared::Type type, + Prepared::Operation operation, + std::vector operands) -> Prepared::Instruction + { + return { Prepared::Instruction::Kind::Bind, + std::move(destination), + std::move(type), + false, + operation, + std::move(operands), + {}, + {} }; + } + + [[nodiscard]] auto + Jump(Prepared::BlockId target) -> Prepared::Terminator + { + return { Prepared::Terminator::Kind::Jump, {}, target, 0U }; + } + + [[nodiscard]] auto + Branch(Prepared::Atom condition, + Prepared::BlockId whenTrue, + Prepared::BlockId whenFalse) -> Prepared::Terminator + { + return { Prepared::Terminator::Kind::Branch, + std::move(condition), + whenTrue, + whenFalse }; + } + + [[nodiscard]] auto + Return(Prepared::Atom value) -> Prepared::Terminator + { + return { Prepared::Terminator::Kind::Return, std::move(value), 0U, 0U }; + } + + /** + * `while (value < 3) { value = value + 1 } return value`, with caller + * chosen block identities, temporary identities and block order. + */ + [[nodiscard]] auto + Loop(Prepared::BlockId header, + Prepared::BlockId body, + Prepared::BlockId exit, + std::uint64_t condition, + std::uint64_t sum, + bool reversed) -> Prepared::CorePrepModule + { + const auto value = [] { + return Read(2U, U"value", Prepared::Type::int64()); + }; + const auto conditionName + = Symbol(condition, U"$coreprep" + std::u32string(1U, U'0')); + const auto sumName + = Symbol(sum, U"$coreprep" + std::u32string(2U, U'9')); + std::vector blocks{ + { 0U, + { Bind(Symbol(2U, U"value"), + Prepared::Type::int64(), + Prepared::Operation::Copy, + { Number(0) }) }, + Jump(header) }, + { header, + { Bind(conditionName, + Prepared::Type::boolean(), + Prepared::Operation::LessThan, + { value(), Number(3) }) }, + Branch(Prepared::Atom::variable(conditionName, + Prepared::Type::boolean()), + body, + exit) }, + { body, + { Bind(sumName, + Prepared::Type::int64(), + Prepared::Operation::Add, + { value(), Number(1) }), + { Prepared::Instruction::Kind::Assign, + Symbol(2U, U"value"), + Prepared::Type::int64(), + false, + Prepared::Operation::Copy, + { Prepared::Atom::variable(sumName, + Prepared::Type::int64()) }, + {}, + {} } }, + Jump(header) }, + { exit, {}, Return(value()) }, + }; + if (reversed) + std::swap(blocks[1], blocks[3]); + return { { U"Parity" }, + { Prepared::Function{ Symbol(1U, U"Evaluate"), + {}, + Prepared::Type::int64(), + 0U, + std::move(blocks) } } }; + } + + [[nodiscard]] auto + Reference() -> Prepared::CorePrepModule + { + return Loop(1U, 2U, 3U, 10U, 11U, false); + } +} // namespace + +TEST_CASE("parity ignores block identities, block order and temporary " + "identities", + "[fuzzing][parity]") +{ + CHECK_FALSE(CompareCorePrep(Reference(), Reference())); + // Different block numbers, a different emission order, and temporaries + // numbered in the opposite order describe the same program. + CHECK_FALSE(CompareCorePrep(Reference(), Loop(7U, 4U, 9U, 31U, 30U, true))); +} + +TEST_CASE("parity ignores blocks that cannot be reached from the entry", + "[fuzzing][parity]") +{ + auto withDeadBlock = Reference(); + withDeadBlock.functions.front().blocks.push_back( + { 40U, + { Bind(Symbol(99U, U"$coreprep99"), + Prepared::Type::int64(), + Prepared::Operation::Copy, + { Number(5) }) }, + Jump(1U) }); + CHECK_FALSE(CompareCorePrep(Reference(), withDeadBlock)); +} + +TEST_CASE("parity reports a back-edge that targets the wrong block", + "[fuzzing][parity]") +{ + // The defect class this oracle exists for: the body jumps to itself + // instead of the loop header. Every block is still well formed. + auto selfLoop = Reference(); + selfLoop.functions.front().blocks[2].terminator = Jump(2U); + const auto difference = CompareCorePrep(Reference(), selfLoop); + REQUIRE(difference); + CHECK(difference->find("Evaluate") != std::string::npos); + CHECK(difference->find("terminator") != std::string::npos); +} + +TEST_CASE("parity distinguishes the true edge from the false edge", + "[fuzzing][parity]") +{ + auto swapped = Reference(); + auto &terminator = swapped.functions.front().blocks[1].terminator; + std::swap(terminator.true_target, terminator.false_target); + CHECK(CompareCorePrep(Reference(), swapped)); +} + +TEST_CASE("parity compares operations, operands, literals and types exactly", + "[fuzzing][parity]") +{ + SECTION("operation") + { + auto changed = Reference(); + changed.functions.front().blocks[1].instructions[0].operation + = Prepared::Operation::LessEqual; + CHECK(CompareCorePrep(Reference(), changed)); + } + SECTION("literal") + { + auto changed = Reference(); + changed.functions.front().blocks[1].instructions[0].operands[1] + = Number(4); + CHECK(CompareCorePrep(Reference(), changed)); + } + SECTION("operand order") + { + auto changed = Reference(); + auto &operands + = changed.functions.front().blocks[2].instructions[0].operands; + std::swap(operands[0], operands[1]); + CHECK(CompareCorePrep(Reference(), changed)); + } + SECTION("type") + { + auto changed = Reference(); + changed.functions.front().blocks[2].instructions[0].type + = Prepared::Type::int32(); + CHECK(CompareCorePrep(Reference(), changed)); + } + SECTION("instruction order") + { + auto changed = Reference(); + auto &instructions = changed.functions.front().blocks[2].instructions; + std::swap(instructions[0], instructions[1]); + CHECK(CompareCorePrep(Reference(), changed)); + } + SECTION("missing instruction") + { + auto changed = Reference(); + changed.functions.front().blocks[2].instructions.pop_back(); + CHECK(CompareCorePrep(Reference(), changed)); + } +} + +TEST_CASE("parity never renames source symbols or temporary kinds", + "[fuzzing][parity]") +{ + SECTION("source symbol identity") + { + auto changed = Reference(); + changed.functions.front().blocks[3].terminator + = Return(Read(5U, U"value", Prepared::Type::int64())); + CHECK(CompareCorePrep(Reference(), changed)); + } + SECTION("source symbol spelling") + { + auto changed = Reference(); + changed.functions.front().symbol.spelling = U"Other"; + CHECK(CompareCorePrep(Reference(), changed)); + } + SECTION("temporary kind") + { + // A condition slot and a plain temporary are different generated + // symbols even when they hold the same value. + auto changed = Reference(); + auto &header = changed.functions.front().blocks[1]; + const auto renamed = Symbol(10U, U"$condition10"); + header.instructions[0].destination = renamed; + header.terminator.value + = Prepared::Atom::variable(renamed, Prepared::Type::boolean()); + CHECK(CompareCorePrep(Reference(), changed)); + } + SECTION("temporary reuse") + { + // Reading the condition temporary where the sum belongs changes + // which value flows, although the instruction shapes are equal. + auto changed = Reference(); + changed.functions.front().blocks[2].instructions[1].operands[0] + = Read(10U, U"$coreprep0", Prepared::Type::int64()); + CHECK(CompareCorePrep(Reference(), changed)); + } +} + +TEST_CASE("parity compares function count and module identity", + "[fuzzing][parity]") +{ + auto extra = Reference(); + extra.functions.push_back(extra.functions.front()); + extra.functions.back().symbol = Symbol(50U, U"Second"); + const auto difference = CompareCorePrep(Reference(), extra); + REQUIRE(difference); + CHECK(difference->find("function count") != std::string::npos); + + auto renamed = Reference(); + renamed.name = { U"Elsewhere" }; + CHECK(CompareCorePrep(Reference(), renamed)); +} diff --git a/Compiler/Haskell/Driver/src/Visual/XSharp/Driver/FFI.hs b/Compiler/Haskell/Driver/src/Visual/XSharp/Driver/FFI.hs index c655e9d8..ca69da46 100644 --- a/Compiler/Haskell/Driver/src/Visual/XSharp/Driver/FFI.hs +++ b/Compiler/Haskell/Driver/src/Visual/XSharp/Driver/FFI.hs @@ -25,6 +25,7 @@ import Foreign.Ptr (FunPtr, Ptr, castPtr, nullPtr) import System.Environment (lookupEnv) import Visual.XSharp.AST import Visual.XSharp.Compiler +import Visual.XSharp.Core.CorePrep.Wire (encodeCorePrep) import Visual.XSharp.Core.Wire import Visual.XSharp.Diagnostic import Visual.XSharp.Diagnostic.SideChannel @@ -128,8 +129,37 @@ frontendFuzzSyntax stage sourcePointer sourceSize _ <- evaluate (length (show (analyzeSyntax input))) pure (CInt 0) +{- | Testing entry: compile like the in-memory source route and additionally +hand out the frontend's own CorePrep lowering of the same optimized Core. The +native pipeline derives CorePrep from Core with its own adapter; delivering both +artifacts of one compilation lets a harness compare the two lowerings instead +of trusting either. Core is delivered first and CorePrep second; a rejected +callback stops the sequence. Production entries never emit CorePrep. +-} frontendFuzzCompile :: Ptr Word8 -> CSize -> FunPtr OutputCallback -> Ptr () -> IO CInt -frontendFuzzCompile = frontendCompileSource +frontendFuzzCompile sourcePointer sourceSize callback context = + withCaughtFailure callback context $ do + copied <- readSource sourcePointer sourceSize + case copied of + Left message -> emitResult callback context (CInt 2) message + Right source -> case Text.decodeUtf8' source of + Left _ -> emitResult callback context (CInt 1) "source is not valid UTF-8" + Right decoded -> + case compileToCorePrep (CompilerInput ".vxs" (Text.unpack decoded)) of + Left diagnostics -> emitBytesResult callback context (CInt 1) 3 (renderDiagnostics diagnostics) + Right artifacts -> + case ( encodeCore defaultCoreWireLimits (artifactOptimizedCore artifacts) + , encodeCorePrep (artifactCorePrep artifacts) + ) of + (Left issue, _) -> emitBytesResult callback context (CInt 3) 3 (show issue) + (_, Left issue) -> emitBytesResult callback context (CInt 3) 3 (show issue) + (Right core, Right corePrep) -> do + coreDelivered <- emitBytes callback context 0 (ByteString.pack core) + prepDelivered <- + if coreDelivered + then emitBytes callback context 4 (ByteString.pack corePrep) + else pure False + pure (if prepDelivered then CInt 0 else CInt 4) emitResult :: FunPtr OutputCallback -> Ptr () -> CInt -> String -> IO CInt emitResult callback context status message = diff --git a/Compiler/Headers/Visual/XSharp/Frontend.h b/Compiler/Headers/Visual/XSharp/Frontend.h index 58909c41..40a1dbfe 100644 --- a/Compiler/Headers/Visual/XSharp/Frontend.h +++ b/Compiler/Headers/Visual/XSharp/Frontend.h @@ -41,7 +41,9 @@ extern "C" /** Structured diagnostics encoded by the diagnostic wire protocol. */ VXS_FRONTEND_DIAGNOSTIC_WIRE = 2, /** Human-readable UTF-8 failure text. */ - VXS_FRONTEND_ERROR_TEXT = 3 + VXS_FRONTEND_ERROR_TEXT = 3, + /** The frontend's own CorePrep lowering; testing entries only. */ + VXS_FRONTEND_COREPREP_WIRE = 4 }; /** @brief Stable result codes returned by frontend ABI entry points. */ @@ -139,16 +141,21 @@ extern "C" const uint8_t *source, size_t source_size); - /** @brief Parse and compile a source fuzz input, delivering verified Core. + /** @brief Parse and compile a source fuzz input, delivering verified Core + * and the frontend's CorePrep lowering of that Core. * * Lexical, syntax, and semantic diagnostics are expected for arbitrary fuzz - * bytes. The native consumer receives Core only through @p output. + * bytes. On success this testing entry invokes @p output twice: first with + * `VXS_FRONTEND_CORE_WIRE`, then with `VXS_FRONTEND_COREPREP_WIRE`. Both + * buffers come from one compilation, so a harness can compare the native + * Core-to-CorePrep adapter with the frontend's lowering. Every other entry + * point delivers exactly one buffer and never emits CorePrep. * @param source Candidate source bytes; malformed UTF-8 is permitted. * @param source_size Number of bytes in @p source. - * @param output Synchronous receiver for a successful Core wire buffer. + * @param output Synchronous receiver for the Core and CorePrep buffers. * @param context Opaque state passed to @p output. - * @return `VXS_FRONTEND_DIAGNOSTICS` for rejected source, zero on Core - * output, or a negative value for an internal/ABI failure. + * @return `VXS_FRONTEND_DIAGNOSTICS` for rejected source, zero when both + * buffers were delivered, or another `vxs_frontend_status` on failure. */ int32_t vxs_frontend_fuzz_compile(const uint8_t *source, diff --git a/Documents/CORE-IR.md b/Documents/CORE-IR.md index f719d92e..472b93f4 100644 --- a/Documents/CORE-IR.md +++ b/Documents/CORE-IR.md @@ -242,7 +242,10 @@ the verified structure and materializes loop headers, exits, latches, and the distinct `for` update block. The Haskell CorePrep lowering and the native Core-to-CorePrep adapter must -build the same loop shape: +build the same program. The source fuzz harness enforces this for every +accepted source by comparing both lowerings of one compilation in a canonical +form; see [Fuzzing](FUZZING.md#coreprep-parity). In particular they must build +the same loop shape: - a `while` or `for` condition owns a dedicated header block. The statements that precede the loop stay in the incoming block, which jumps to the header diff --git a/Documents/FUZZING.md b/Documents/FUZZING.md index 81d6075f..49ca0b6e 100644 --- a/Documents/FUZZING.md +++ b/Documents/FUZZING.md @@ -38,6 +38,29 @@ subset; it is not an oracle for arbitrary Visual X# programs. Invalid source is a normal rejection, whereas internal failures and verified-model inconsistencies fail the campaign. +### CorePrep parity + +The frontend lowers its optimized Core to CorePrep, and the native pipeline +lowers the same Core again with its own adapter. Only the native result +reaches Xpp, and both lowerings are well formed, so a divergence is a +miscompile that no later verifier can see. Every source that the frontend +accepts in `source_llvm_fuzzer`, `differential_fuzzer` and `source_fuzz_smoke` +is therefore checked for parity: the testing entry of the frontend delivers +the Core and CorePrep buffers of one compilation, the harness runs the native +adapter on that Core, and the two CorePrep modules must be structurally equal. + +Both modules are compared in a canonical form. Reachable blocks are ordered +depth first from the entry, true edge before false edge, and `$`-prefixed +generated symbols are renamed in first-use order while keeping their kind. +Source symbols, types, literals, operations, operand and instruction order, +and edge roles are compared exactly; blocks unreachable from the entry are +ignored. `Compiler/Fuzzing/Tests/coreprep_parity_tests` pins what the +comparison may and may not ignore. Reintroducing the for-loop update defect +makes the smoke fail in this check before any generated code runs. + +Parity proves that the two lowerings agree, not that either is correct. The +executable oracle below remains the check against an independent expectation. + Arbitrary source and generated arithmetic use separate corpora and equal per-target time budgets. This lets source mutations reach native lowering without repeatedly creating two ORC sessions for unrelated generated code. @@ -213,6 +236,9 @@ The harnesses above are evidence about the inputs they ran, not proofs: - HPC feedback is expression-tick coverage of the frontend, without memory-safety instrumentation, and its mutator is byte- and token-level, not grammar-aware; +- CorePrep parity only covers programs the mutators and the fixed smoke + sources reach. Closures and templates have no deterministic parity source + yet, and parity cannot detect a defect both lowerings share; - prebuilt LLVM and the GHC runtime are not instrumented by any campaign. Expand these deliberately instead of equating a green workflow with completion diff --git a/Documents/TESTING.md b/Documents/TESTING.md index f35a3205..1b1b5b77 100644 --- a/Documents/TESTING.md +++ b/Documents/TESTING.md @@ -118,7 +118,7 @@ go run ./helpers/cmd/develop doctor go run ./helpers/cmd/develop test ``` -The command executes 19 Catch3 binaries, one C11 ABI contract executable and +The command executes 20 Catch3 binaries, one C11 ABI contract executable and one source/differential fuzz smoke executable on Windows 10/11, macOS Sequoia/Tahoe, Ubuntu 26.04 LTS, and Fedora 43. Bazel selects the host configuration automatically; no public test instruction diff --git a/helpers/internal/development/build.go b/helpers/internal/development/build.go index 2750aec6..8992d289 100644 --- a/helpers/internal/development/build.go +++ b/helpers/internal/development/build.go @@ -33,6 +33,7 @@ var nativeTargets = []string{ "//Compiler/Runtime/AARC/Tests:aarc_runtime_tests", "//Compiler/Runtime/AARC/Tests:aarc_c_abi_tests", "//Compiler/Fuzzing:source_fuzz_smoke", + "//Compiler/Fuzzing/Tests:coreprep_parity_tests", "//Compiler/ProjectSystem/Bridge/Tests:project_registry_tests", "//Interactive/Tests:interactive_tests", }