diff --git a/Compiler/Cli/Commands/Frontend.cpp b/Compiler/Cli/Commands/Frontend.cpp index 81e8b29f..d60b4b14 100644 --- a/Compiler/Cli/Commands/Frontend.cpp +++ b/Compiler/Cli/Commands/Frontend.cpp @@ -59,6 +59,11 @@ namespace Visual::XSharp::Cli::Frontend std::optional kind; std::vector bytes; std::string error; + // Only the testing route sets this. It lets one CorePrep payload + // follow a Core payload; every other route keeps the + // single-payload contract. + bool acceptCorePrep{}; + std::optional> corePrep; }; auto @@ -313,10 +318,26 @@ namespace Visual::XSharp::Cli::Frontend const std::uint8_t *bytes, std::size_t size) noexcept -> std::int32_t { - if (context == nullptr || rawKind > 3U + if (context == nullptr || rawKind > 4U || (size != 0U && bytes == nullptr)) return 1; auto &captured = *static_cast(context); + if (static_cast(rawKind) == OutputKind::CorePrepWire) + { + // CorePrep is accepted once, only after Core, and only on + // the route that asked for it. + if (!captured.acceptCorePrep + || captured.kind != OutputKind::CoreWire + || captured.corePrep.has_value() + || size > kMaximumCoreBytes) + { + captured.error = "frontend emitted an unexpected CorePrep " + "payload"; + return 1; + } + captured.corePrep.emplace(bytes, bytes + size); + return 0; + } if (captured.kind.has_value()) { captured.error @@ -450,8 +471,39 @@ namespace Visual::XSharp::Cli::Frontend {}, frontend.Error() }; CapturedOutput output; + output.acceptCorePrep = true; const auto status = frontend.FuzzCompile(source.data(), source.size(), output); return MakeResult(status, std::move(output)); } + + auto + FuzzCompileStages(std::span source) -> StageResult + { + auto &frontend = GetFrontend(); + if (!frontend.IsReady()) + return { { Status::InternalError, + OutputKind::ErrorText, + {}, + frontend.Error() }, + {} }; + CapturedOutput output; + output.acceptCorePrep = true; + const auto status + = frontend.FuzzCompile(source.data(), source.size(), output); + auto corePrep = std::move(output.corePrep); + StageResult result{ MakeResult(status, std::move(output)), {} }; + if (result.core.succeeded() && result.core.kind == OutputKind::CoreWire) + { + if (!corePrep) + { + result.core.status = Status::InternalError; + result.core.error + = "frontend delivered Core without its CorePrep lowering"; + return result; + } + result.corePrep = std::move(*corePrep); + } + return result; + } } // namespace Visual::XSharp::Cli::Frontend diff --git a/Compiler/Cli/Commands/Frontend.hpp b/Compiler/Cli/Commands/Frontend.hpp index ee0ae730..7271f4ef 100644 --- a/Compiler/Cli/Commands/Frontend.hpp +++ b/Compiler/Cli/Commands/Frontend.hpp @@ -19,7 +19,8 @@ namespace Visual::XSharp::Cli::Frontend CoreWire = 0, ///< Versioned, verified Core wire bytes. ProjectSourceList = 1, ///< NUL-delimited UTF-8 source paths. DiagnosticWire = 2, ///< Structured source diagnostics. - ErrorText = 3 ///< Human-readable UTF-8 failure text. + ErrorText = 3, ///< Human-readable UTF-8 failure text. + CorePrepWire = 4 ///< Frontend CorePrep lowering; testing only. }; /// @brief Stable operation outcomes returned by the C ABI. @@ -79,4 +80,23 @@ namespace Visual::XSharp::Cli::Frontend /// failures. [[nodiscard]] auto FuzzCompile(std::span source) -> Result; + + /// @brief Core and the frontend's own CorePrep from one compilation. + struct StageResult final + { + Result core; ///< Status and Core wire, as returned by FuzzCompile. + /// Frontend-lowered CorePrep wire; empty unless core succeeded. + std::vector corePrep; + }; + + /// @brief Compile source bytes and keep both frontend artifacts. + /// + /// The native pipeline lowers Core to CorePrep with its own adapter. This + /// testing route additionally returns the Haskell lowering of the same + /// Core so the two can be compared. + /// @param source Candidate source bytes; malformed input is permitted. + /// @return Owned Core and CorePrep buffers; a successful status without + /// CorePrep is reported as an internal error. + [[nodiscard]] auto + FuzzCompileStages(std::span source) -> StageResult; } // namespace Visual::XSharp::Cli::Frontend diff --git a/Compiler/Fuzzing/BUILD.bazel b/Compiler/Fuzzing/BUILD.bazel index 0d366e82..14278ddc 100644 --- a/Compiler/Fuzzing/BUILD.bazel +++ b/Compiler/Fuzzing/BUILD.bazel @@ -33,14 +33,28 @@ cc_binary( deps = [":wire_fuzz_harness"], ) +# Structural comparison of two CorePrep lowerings. It has no frontend or LLVM +# dependency so its own contract can be tested without either. +cc_library( + name = "coreprep_parity", + srcs = ["CorePrepParity.cpp"], + hdrs = ["CorePrepParity.hpp"], + visibility = ["//Compiler/Fuzzing/Tests:__pkg__"], + deps = ["//Compiler/Headers/Visual/XSharp/Core:core"], +) + cc_library( name = "source_fuzz_harness", srcs = ["SourceFuzz.cpp"], hdrs = ["SourceFuzz.hpp"], deps = [ + ":coreprep_parity", "//Compiler/Backend/LLVM:llvm_backend", "//Compiler/Cli/Commands:frontend", + "//Compiler/Core:core", "//Compiler/Driver:pipeline", + "//Compiler/Headers/Visual/XSharp/Core:core", + "//Compiler/Headers/Visual/XSharp/Core:coreprep_wire", ], ) diff --git a/Compiler/Fuzzing/CorePrepParity.cpp b/Compiler/Fuzzing/CorePrepParity.cpp new file mode 100644 index 00000000..44b907c8 --- /dev/null +++ b/Compiler/Fuzzing/CorePrepParity.cpp @@ -0,0 +1,301 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include + +#include "CorePrepParity.hpp" + +namespace Visual::XSharp::Fuzzing +{ + namespace + { + namespace Prepared = ::visual_xsharp::core; + + // Canonical identities start far above any real symbol so a renamed + // generated symbol can never equal a source symbol. + constexpr Prepared::SymbolId kCanonicalBase = Prepared::SymbolId{ 1 } + << 48U; + + /// Generated symbols are spelled with a leading `$`, which cannot + /// occur in a source identifier. + [[nodiscard]] auto + IsGenerated(const Prepared::SymbolName &symbol) -> bool + { + return !symbol.spelling.empty() && symbol.spelling.front() == U'$'; + } + + /// `$coreprep17` and `$coreprep4` are the same kind of temporary; + /// the digits only repeat the identity. + [[nodiscard]] auto + KindOf(const std::u32string &spelling) -> std::u32string + { + auto end = spelling.size(); + while (end > 0U && spelling[end - 1U] >= U'0' + && spelling[end - 1U] <= U'9') + --end; + return spelling.substr(0U, end); + } + + /** + * @brief Renames generated symbols in first-use order. + * + * Both lowerings may number temporaries differently, and one module + * shares one numbering because a lifted closure is named in its + * parent and defined later. Source symbols are never renamed. + */ + class Renamer final + { + public: + void + Apply(Prepared::SymbolName &symbol) + { + if (!IsGenerated(symbol)) + return; + const auto [found, inserted] + = names_.try_emplace(symbol.id, Prepared::SymbolName{}); + if (inserted) + found->second = { kCanonicalBase + names_.size(), + KindOf(symbol.spelling) }; + symbol = found->second; + } + + void + Apply(Prepared::Atom &atom) + { + if (atom.kind == Prepared::Atom::Kind::Variable) + Apply(atom.symbol); + } + + private: + std::unordered_map names_; + }; + + [[nodiscard]] auto + Successors(const Prepared::Block &block) + -> std::vector + { + switch (block.terminator.kind) + { + case Prepared::Terminator::Kind::Jump: + return { block.terminator.true_target }; + case Prepared::Terminator::Kind::Branch: + return { block.terminator.true_target, + block.terminator.false_target }; + case Prepared::Terminator::Kind::Return: + case Prepared::Terminator::Kind::Unreachable: + break; + } + return {}; + } + + /** + * @brief Reorders reachable blocks in depth-first preorder. + * + * Block identities are allocation artifacts. Following the true + * edge before the false edge from the entry gives both lowerings + * the same order exactly when their control-flow graphs are + * isomorphic with matching edge roles. Blocks that cannot be + * reached from the entry, such as the join after two returning + * branches, never execute and are left out. + */ + void + CanonicalizeBlocks(Prepared::Function &function) + { + std::unordered_map position; + for (std::size_t index = 0U; index < function.blocks.size(); + ++index) + position.emplace(function.blocks[index].id, index); + + std::unordered_map renumbered; + std::vector order; + std::vector pending{ function.entry }; + while (!pending.empty()) + { + const auto id = pending.back(); + pending.pop_back(); + const auto found = position.find(id); + if (found == position.end() || renumbered.contains(id)) + continue; + renumbered.emplace( + id, + static_cast(order.size())); + order.push_back(found->second); + auto successors = Successors(function.blocks[found->second]); + // The stack reverses order, so push the false edge first. + std::ranges::reverse(successors); + pending.insert(pending.end(), + successors.begin(), + successors.end()); + } + + const auto target = [&renumbered](Prepared::BlockId id) { + const auto found = renumbered.find(id); + // A dangling target is a verifier matter; keep it distinct + // from every canonical block. + return found == renumbered.end() ? ~Prepared::BlockId{} + : found->second; + }; + std::vector blocks; + blocks.reserve(order.size()); + for (const auto index : order) + { + auto block = std::move(function.blocks[index]); + block.id = static_cast(blocks.size()); + switch (block.terminator.kind) + { + case Prepared::Terminator::Kind::Branch: + block.terminator.false_target + = target(block.terminator.false_target); + [[fallthrough]]; + case Prepared::Terminator::Kind::Jump: + block.terminator.true_target + = target(block.terminator.true_target); + break; + case Prepared::Terminator::Kind::Return: + case Prepared::Terminator::Kind::Unreachable: + break; + } + blocks.push_back(std::move(block)); + } + function.blocks = std::move(blocks); + function.entry = 0U; + } + + [[nodiscard]] auto + Canonicalize(Prepared::CorePrepModule module) + -> Prepared::CorePrepModule + { + Renamer renamer; + for (auto &function : module.functions) + { + CanonicalizeBlocks(function); + renamer.Apply(function.symbol); + for (auto ¶meter : function.parameters) + renamer.Apply(parameter.symbol); + for (auto &block : function.blocks) + { + for (auto &instruction : block.instructions) + { + // Operands are read before the destination is + // written, so name them first. + for (auto &operand : instruction.operands) + renamer.Apply(operand); + for (auto &capture : instruction.captures) + { + renamer.Apply(capture.value); + renamer.Apply(capture.symbol); + } + renamer.Apply(instruction.closure_function); + renamer.Apply(instruction.destination); + } + renamer.Apply(block.terminator.value); + } + } + return module; + } + + [[nodiscard]] auto + Narrow(const std::u32string &text) -> std::string + { + std::string result; + for (const auto point : text) + result.push_back(point < 0x80U ? static_cast(point) + : '?'); + return result; + } + + [[nodiscard]] auto + Describe(const Prepared::Instruction &instruction) -> std::string + { + return "kind " + + std::to_string(static_cast(instruction.kind)) + + ", operation " + + std::to_string( + static_cast(instruction.operation)) + + ", destination " + Narrow(instruction.destination.spelling) + + ", " + std::to_string(instruction.operands.size()) + + " operand(s)"; + } + + [[nodiscard]] auto + Describe(const Prepared::Terminator &terminator) -> std::string + { + return "terminator kind " + + std::to_string(static_cast(terminator.kind)) + + " -> " + std::to_string(terminator.true_target) + "/" + + std::to_string(terminator.false_target); + } + + [[nodiscard]] auto + FirstDifference(const Prepared::Function &frontend, + const Prepared::Function &native) -> std::string + { + if (frontend.symbol != native.symbol + || frontend.parameters != native.parameters + || frontend.return_type != native.return_type + || frontend.sourceFile != native.sourceFile) + return "signature or source owner differs"; + if (frontend.blocks.size() != native.blocks.size()) + return "reachable block count " + + std::to_string(frontend.blocks.size()) + + " (frontend) versus " + + std::to_string(native.blocks.size()) + " (native)"; + for (std::size_t block = 0U; block < frontend.blocks.size(); + ++block) + { + const auto &left = frontend.blocks[block]; + const auto &right = native.blocks[block]; + const auto where = "canonical block " + std::to_string(block); + const auto shared = std::min(left.instructions.size(), + right.instructions.size()); + for (std::size_t index = 0U; index < shared; ++index) + if (left.instructions[index] != right.instructions[index]) + return where + ", instruction " + std::to_string(index) + + ": frontend {" + + Describe(left.instructions[index]) + + "} versus native {" + + Describe(right.instructions[index]) + "}"; + if (left.instructions.size() != right.instructions.size()) + return where + ": " + + std::to_string(left.instructions.size()) + + " instruction(s) (frontend) versus " + + std::to_string(right.instructions.size()) + + " (native)"; + if (left.terminator != right.terminator) + return where + ": frontend {" + Describe(left.terminator) + + "} versus native {" + Describe(right.terminator) + + "}"; + } + return "functions differ outside their blocks"; + } + } // namespace + + auto + CompareCorePrep(const Prepared::CorePrepModule &frontend, + const Prepared::CorePrepModule &native) + -> std::optional + { + const auto left = Canonicalize(frontend); + const auto right = Canonicalize(native); + if (left == right) + return std::nullopt; + if (left.name != right.name || left.sourceFiles != right.sourceFiles) + return "module name or source catalog differs"; + if (left.functions.size() != right.functions.size()) + return "function count " + std::to_string(left.functions.size()) + + " (frontend) versus " + + std::to_string(right.functions.size()) + " (native)"; + for (std::size_t index = 0U; index < left.functions.size(); ++index) + if (left.functions[index] != right.functions[index]) + return "function " + std::to_string(index) + " (" + + Narrow(left.functions[index].symbol.spelling) + "): " + + FirstDifference(left.functions[index], + right.functions[index]); + return "modules differ"; + } +} // namespace Visual::XSharp::Fuzzing diff --git a/Compiler/Fuzzing/CorePrepParity.hpp b/Compiler/Fuzzing/CorePrepParity.hpp new file mode 100644 index 00000000..9738cac5 --- /dev/null +++ b/Compiler/Fuzzing/CorePrepParity.hpp @@ -0,0 +1,34 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +#pragma once + +#include +#include + +#include "Visual/XSharp/Core/CorePrep.hpp" + +namespace Visual::XSharp::Fuzzing +{ + /** + * @brief Compare two CorePrep lowerings of the same Core module. + * + * The Haskell frontend and the native Core-to-CorePrep adapter must + * build the same program: the same reachable control-flow graph with the + * same instructions in every block. Block identities and the identities + * of generated symbols are allocation details, so both modules are first + * brought to a canonical form: reachable blocks in depth-first preorder + * from the entry, true edge before false edge, and `$`-prefixed generated + * symbols renamed in first-use order while keeping their kind. Source + * symbols, types, literals, operations, operand order, instruction order + * and edge roles are compared exactly. Blocks unreachable from the entry + * are ignored. + * + * @param frontend CorePrep produced by the Haskell frontend. + * @param native CorePrep produced by the native adapter. + * @return std::nullopt when equivalent, otherwise the first difference. + */ + [[nodiscard]] auto + CompareCorePrep(const ::visual_xsharp::core::CorePrepModule &frontend, + const ::visual_xsharp::core::CorePrepModule &native) + -> std::optional; +} // namespace Visual::XSharp::Fuzzing diff --git a/Compiler/Fuzzing/SourceFuzz.cpp b/Compiler/Fuzzing/SourceFuzz.cpp index 009250b8..ed862872 100644 --- a/Compiler/Fuzzing/SourceFuzz.cpp +++ b/Compiler/Fuzzing/SourceFuzz.cpp @@ -11,6 +11,7 @@ #include #include "Compiler/Cli/Commands/Frontend.hpp" +#include "CorePrepParity.hpp" #include "SourceFuzz.hpp" #include "Visual/XSharp/Backend/LLVM.hpp" #include "Visual/XSharp/Pipeline.hpp" @@ -167,6 +168,17 @@ namespace Visual::XSharp::Fuzzing "}\n"; } + /** + * @brief Compile once and require both CorePrep lowerings to agree. + * + * The frontend lowers its optimized Core to CorePrep, and the native + * pipeline lowers the same Core again with its own adapter. Only + * the native result reaches Xpp, so a divergence is a miscompile + * that no later verifier can see: both lowerings are well formed. + * This check decodes the Core and CorePrep buffers of one + * compilation, runs the native adapter, and fails on the first + * structural difference. + */ [[nodiscard]] auto CompileSource(std::span source) -> Frontend::Result { @@ -175,7 +187,34 @@ namespace Visual::XSharp::Fuzzing Frontend::OutputKind::ErrorText, {}, "source fuzz input exceeds 64 KiB" }; - return Frontend::FuzzCompile(source); + auto stages = Frontend::FuzzCompileStages(source); + if (!stages.core.succeeded() + || stages.core.kind != Frontend::OutputKind::CoreWire) + return std::move(stages.core); + + const auto core = Core::Wire::Decode(stages.core.bytes); + if (!core) + llvm::report_fatal_error(llvm::Twine( + "native Core reader rejected frontend Core wire")); + // Unverified Core is rejected by the pipeline with its own + // report; lowering it here would compare undefined shapes. + if (!Core::Verify(*core.module).empty()) + return std::move(stages.core); + const auto frontendCorePrep + = ::visual_xsharp::core::wire::decode(stages.corePrep); + if (!frontendCorePrep) + llvm::report_fatal_error(llvm::Twine( + "native CorePrep reader rejected frontend CorePrep wire")); + const auto difference + = CompareCorePrep(*frontendCorePrep.module, + Core::CorePrep::Prepare(*core.module)); + if (difference) + llvm::report_fatal_error(llvm::Twine( + "frontend and native CorePrep lowerings differ: " + + *difference + "; source:\n" + + std::string(reinterpret_cast(source.data()), + source.size()))); + return std::move(stages.core); } [[nodiscard]] auto @@ -352,6 +391,13 @@ namespace Visual::XSharp::Fuzzing (void)ConsumeVerifiedCore(compiled, "arbitrary source fuzz input"); } + void + ExerciseAcceptedSource(std::span input) + { + const auto compiled = CompileSource(input); + (void)ConsumeVerifiedCore(compiled, "source that must be accepted"); + } + void ExerciseDifferentialOracle(std::span input) { diff --git a/Compiler/Fuzzing/SourceFuzz.hpp b/Compiler/Fuzzing/SourceFuzz.hpp index ced12136..d86456f2 100644 --- a/Compiler/Fuzzing/SourceFuzz.hpp +++ b/Compiler/Fuzzing/SourceFuzz.hpp @@ -9,7 +9,9 @@ namespace Visual::XSharp::Fuzzing { // These routes deliberately share production Haskell entry points. The // syntax stages stop after the lexer or parser; source compilation goes - // through verified Core, Xpp, Xmm and LLVM lowering. + // through verified Core, Xpp, Xmm and LLVM lowering. Every accepted + // source is also lowered to CorePrep by both the frontend and the native + // adapter, and the two results must be structurally equal. void ExerciseLexer(std::span input); void @@ -18,4 +20,9 @@ namespace Visual::XSharp::Fuzzing ExerciseSourceToLlvm(std::span input); void ExerciseDifferentialOracle(std::span input); + /// Like ExerciseSourceToLlvm, but the source is known to be valid: a + /// frontend rejection is a failure. Deterministic checks use this so a + /// program that silently stopped compiling cannot pass as "rejected". + void + ExerciseAcceptedSource(std::span input); } // namespace Visual::XSharp::Fuzzing diff --git a/Compiler/Fuzzing/SourceFuzzSmoke.cpp b/Compiler/Fuzzing/SourceFuzzSmoke.cpp index 52ae85bb..79d860e7 100644 --- a/Compiler/Fuzzing/SourceFuzzSmoke.cpp +++ b/Compiler/Fuzzing/SourceFuzzSmoke.cpp @@ -24,6 +24,38 @@ main() const std::span emptySource; Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(emptySource); Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(source); + // Control-flow shapes whose frontend and native CorePrep lowerings must + // agree. They combine forms the generated programs below keep separate: + // loops inside loops, short-circuit operators as loop conditions, and + // several functions sharing one module-wide symbol numbering. + constexpr std::array accepted{ + "namespace Parity; class Program { public static int Evaluate() { " + "int total = 0; for (int outer = 0; outer < 4; outer++) { " + "if (outer == 2) { continue; } int inner = 0; " + "while (inner < 3) { if (inner == 1) { inner++; continue; } " + "total = total + outer + inner; inner++; } } return total; } }", + "namespace Parity; class Program { public static int Evaluate() { " + "int index = 0; int total = 0; " + "while (index < 9 && (index < 4 || total < 20)) { " + "total = total + index; index++; } " + "do { total++; } while (total < 40 && index \\= 0); " + "return total; } }", + "namespace Parity; class Program { " + "public static bool Down(_ int n) { return n == 0 || Down(n - 1); } " + "public static int Pick(_ int n) { if (Down(n) && n < 6) { " + "return n; } return 0 - 1; } " + "public static int Evaluate() { return Pick(3) + Pick(8); } }", + "namespace Parity; class First { public static long Evaluate() { " + "long result = 8; if (result > 4) { result = result - 2; } " + "return result; } } class Second { public static long Other() { " + "long value = 3; for (int step = 0; step < 2; step++) { " + "value = value * 2; } return value; } }", + }; + for (const auto text : accepted) + Visual::XSharp::Fuzzing::ExerciseAcceptedSource( + std::span( + reinterpret_cast(text.data()), + text.size())); llvm::errs() << "Differential smoke: mixed seed\n"; Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(expressionSeed); llvm::errs() << "Differential smoke: empty seed\n"; diff --git a/Compiler/Fuzzing/Tests/BUILD.bazel b/Compiler/Fuzzing/Tests/BUILD.bazel new file mode 100644 index 00000000..64fcbd12 --- /dev/null +++ b/Compiler/Fuzzing/Tests/BUILD.bazel @@ -0,0 +1,14 @@ +load("@rules_cc//cc:cc_binary.bzl", "cc_binary") + +package(default_visibility = ["//visibility:private"]) + +cc_binary( + name = "coreprep_parity_tests", + srcs = ["CorePrepParityTests.cpp"], + deps = [ + "//Compiler/Fuzzing:coreprep_parity", + "//Compiler/Headers/Visual/XSharp/Core:core", + "@catch3//:catch2_main", + "@catch3//src/Progmasoft:catch3", + ], +) diff --git a/Compiler/Fuzzing/Tests/CorePrepParityTests.cpp b/Compiler/Fuzzing/Tests/CorePrepParityTests.cpp new file mode 100644 index 00000000..f6fbaeb5 --- /dev/null +++ b/Compiler/Fuzzing/Tests/CorePrepParityTests.cpp @@ -0,0 +1,295 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include + +#include "Compiler/Fuzzing/CorePrepParity.hpp" +#include "Visual/XSharp/Core/CorePrep.hpp" + +// The parity oracle is only as strong as its notion of "same program". These +// tests pin what it must ignore (block and temporary numbering, unreachable +// blocks) and, more importantly, what it must never ignore. + +namespace +{ + namespace Prepared = visual_xsharp::core; + using Visual::XSharp::Fuzzing::CompareCorePrep; + + [[nodiscard]] auto + Symbol(std::uint64_t id, std::u32string spelling) -> Prepared::SymbolName + { + return { id, std::move(spelling) }; + } + + [[nodiscard]] auto + Read(std::uint64_t id, std::u32string spelling, Prepared::Type type) + -> Prepared::Atom + { + return Prepared::Atom::variable(Symbol(id, std::move(spelling)), + std::move(type)); + } + + [[nodiscard]] auto + Number(std::int64_t value) -> Prepared::Atom + { + return Prepared::Atom::constant(value, Prepared::Type::int64()); + } + + [[nodiscard]] auto + Bind(Prepared::SymbolName destination, + Prepared::Type type, + Prepared::Operation operation, + std::vector operands) -> Prepared::Instruction + { + return { Prepared::Instruction::Kind::Bind, + std::move(destination), + std::move(type), + false, + operation, + std::move(operands), + {}, + {} }; + } + + [[nodiscard]] auto + Jump(Prepared::BlockId target) -> Prepared::Terminator + { + return { Prepared::Terminator::Kind::Jump, {}, target, 0U }; + } + + [[nodiscard]] auto + Branch(Prepared::Atom condition, + Prepared::BlockId whenTrue, + Prepared::BlockId whenFalse) -> Prepared::Terminator + { + return { Prepared::Terminator::Kind::Branch, + std::move(condition), + whenTrue, + whenFalse }; + } + + [[nodiscard]] auto + Return(Prepared::Atom value) -> Prepared::Terminator + { + return { Prepared::Terminator::Kind::Return, std::move(value), 0U, 0U }; + } + + /** + * `while (value < 3) { value = value + 1 } return value`, with caller + * chosen block identities, temporary identities and block order. + */ + [[nodiscard]] auto + Loop(Prepared::BlockId header, + Prepared::BlockId body, + Prepared::BlockId exit, + std::uint64_t condition, + std::uint64_t sum, + bool reversed) -> Prepared::CorePrepModule + { + const auto value = [] { + return Read(2U, U"value", Prepared::Type::int64()); + }; + const auto conditionName + = Symbol(condition, U"$coreprep" + std::u32string(1U, U'0')); + const auto sumName + = Symbol(sum, U"$coreprep" + std::u32string(2U, U'9')); + std::vector blocks{ + { 0U, + { Bind(Symbol(2U, U"value"), + Prepared::Type::int64(), + Prepared::Operation::Copy, + { Number(0) }) }, + Jump(header) }, + { header, + { Bind(conditionName, + Prepared::Type::boolean(), + Prepared::Operation::LessThan, + { value(), Number(3) }) }, + Branch(Prepared::Atom::variable(conditionName, + Prepared::Type::boolean()), + body, + exit) }, + { body, + { Bind(sumName, + Prepared::Type::int64(), + Prepared::Operation::Add, + { value(), Number(1) }), + { Prepared::Instruction::Kind::Assign, + Symbol(2U, U"value"), + Prepared::Type::int64(), + false, + Prepared::Operation::Copy, + { Prepared::Atom::variable(sumName, + Prepared::Type::int64()) }, + {}, + {} } }, + Jump(header) }, + { exit, {}, Return(value()) }, + }; + if (reversed) + std::swap(blocks[1], blocks[3]); + return { { U"Parity" }, + { Prepared::Function{ Symbol(1U, U"Evaluate"), + {}, + Prepared::Type::int64(), + 0U, + std::move(blocks) } } }; + } + + [[nodiscard]] auto + Reference() -> Prepared::CorePrepModule + { + return Loop(1U, 2U, 3U, 10U, 11U, false); + } +} // namespace + +TEST_CASE("parity ignores block identities, block order and temporary " + "identities", + "[fuzzing][parity]") +{ + CHECK_FALSE(CompareCorePrep(Reference(), Reference())); + // Different block numbers, a different emission order, and temporaries + // numbered in the opposite order describe the same program. + CHECK_FALSE(CompareCorePrep(Reference(), Loop(7U, 4U, 9U, 31U, 30U, true))); +} + +TEST_CASE("parity ignores blocks that cannot be reached from the entry", + "[fuzzing][parity]") +{ + auto withDeadBlock = Reference(); + withDeadBlock.functions.front().blocks.push_back( + { 40U, + { Bind(Symbol(99U, U"$coreprep99"), + Prepared::Type::int64(), + Prepared::Operation::Copy, + { Number(5) }) }, + Jump(1U) }); + CHECK_FALSE(CompareCorePrep(Reference(), withDeadBlock)); +} + +TEST_CASE("parity reports a back-edge that targets the wrong block", + "[fuzzing][parity]") +{ + // The defect class this oracle exists for: the body jumps to itself + // instead of the loop header. Every block is still well formed. + auto selfLoop = Reference(); + selfLoop.functions.front().blocks[2].terminator = Jump(2U); + const auto difference = CompareCorePrep(Reference(), selfLoop); + REQUIRE(difference); + CHECK(difference->find("Evaluate") != std::string::npos); + CHECK(difference->find("terminator") != std::string::npos); +} + +TEST_CASE("parity distinguishes the true edge from the false edge", + "[fuzzing][parity]") +{ + auto swapped = Reference(); + auto &terminator = swapped.functions.front().blocks[1].terminator; + std::swap(terminator.true_target, terminator.false_target); + CHECK(CompareCorePrep(Reference(), swapped)); +} + +TEST_CASE("parity compares operations, operands, literals and types exactly", + "[fuzzing][parity]") +{ + SECTION("operation") + { + auto changed = Reference(); + changed.functions.front().blocks[1].instructions[0].operation + = Prepared::Operation::LessEqual; + CHECK(CompareCorePrep(Reference(), changed)); + } + SECTION("literal") + { + auto changed = Reference(); + changed.functions.front().blocks[1].instructions[0].operands[1] + = Number(4); + CHECK(CompareCorePrep(Reference(), changed)); + } + SECTION("operand order") + { + auto changed = Reference(); + auto &operands + = changed.functions.front().blocks[2].instructions[0].operands; + std::swap(operands[0], operands[1]); + CHECK(CompareCorePrep(Reference(), changed)); + } + SECTION("type") + { + auto changed = Reference(); + changed.functions.front().blocks[2].instructions[0].type + = Prepared::Type::int32(); + CHECK(CompareCorePrep(Reference(), changed)); + } + SECTION("instruction order") + { + auto changed = Reference(); + auto &instructions = changed.functions.front().blocks[2].instructions; + std::swap(instructions[0], instructions[1]); + CHECK(CompareCorePrep(Reference(), changed)); + } + SECTION("missing instruction") + { + auto changed = Reference(); + changed.functions.front().blocks[2].instructions.pop_back(); + CHECK(CompareCorePrep(Reference(), changed)); + } +} + +TEST_CASE("parity never renames source symbols or temporary kinds", + "[fuzzing][parity]") +{ + SECTION("source symbol identity") + { + auto changed = Reference(); + changed.functions.front().blocks[3].terminator + = Return(Read(5U, U"value", Prepared::Type::int64())); + CHECK(CompareCorePrep(Reference(), changed)); + } + SECTION("source symbol spelling") + { + auto changed = Reference(); + changed.functions.front().symbol.spelling = U"Other"; + CHECK(CompareCorePrep(Reference(), changed)); + } + SECTION("temporary kind") + { + // A condition slot and a plain temporary are different generated + // symbols even when they hold the same value. + auto changed = Reference(); + auto &header = changed.functions.front().blocks[1]; + const auto renamed = Symbol(10U, U"$condition10"); + header.instructions[0].destination = renamed; + header.terminator.value + = Prepared::Atom::variable(renamed, Prepared::Type::boolean()); + CHECK(CompareCorePrep(Reference(), changed)); + } + SECTION("temporary reuse") + { + // Reading the condition temporary where the sum belongs changes + // which value flows, although the instruction shapes are equal. + auto changed = Reference(); + changed.functions.front().blocks[2].instructions[1].operands[0] + = Read(10U, U"$coreprep0", Prepared::Type::int64()); + CHECK(CompareCorePrep(Reference(), changed)); + } +} + +TEST_CASE("parity compares function count and module identity", + "[fuzzing][parity]") +{ + auto extra = Reference(); + extra.functions.push_back(extra.functions.front()); + extra.functions.back().symbol = Symbol(50U, U"Second"); + const auto difference = CompareCorePrep(Reference(), extra); + REQUIRE(difference); + CHECK(difference->find("function count") != std::string::npos); + + auto renamed = Reference(); + renamed.name = { U"Elsewhere" }; + CHECK(CompareCorePrep(Reference(), renamed)); +} diff --git a/Compiler/Haskell/Driver/src/Visual/XSharp/Driver/FFI.hs b/Compiler/Haskell/Driver/src/Visual/XSharp/Driver/FFI.hs index c655e9d8..ca69da46 100644 --- a/Compiler/Haskell/Driver/src/Visual/XSharp/Driver/FFI.hs +++ b/Compiler/Haskell/Driver/src/Visual/XSharp/Driver/FFI.hs @@ -25,6 +25,7 @@ import Foreign.Ptr (FunPtr, Ptr, castPtr, nullPtr) import System.Environment (lookupEnv) import Visual.XSharp.AST import Visual.XSharp.Compiler +import Visual.XSharp.Core.CorePrep.Wire (encodeCorePrep) import Visual.XSharp.Core.Wire import Visual.XSharp.Diagnostic import Visual.XSharp.Diagnostic.SideChannel @@ -128,8 +129,37 @@ frontendFuzzSyntax stage sourcePointer sourceSize _ <- evaluate (length (show (analyzeSyntax input))) pure (CInt 0) +{- | Testing entry: compile like the in-memory source route and additionally +hand out the frontend's own CorePrep lowering of the same optimized Core. The +native pipeline derives CorePrep from Core with its own adapter; delivering both +artifacts of one compilation lets a harness compare the two lowerings instead +of trusting either. Core is delivered first and CorePrep second; a rejected +callback stops the sequence. Production entries never emit CorePrep. +-} frontendFuzzCompile :: Ptr Word8 -> CSize -> FunPtr OutputCallback -> Ptr () -> IO CInt -frontendFuzzCompile = frontendCompileSource +frontendFuzzCompile sourcePointer sourceSize callback context = + withCaughtFailure callback context $ do + copied <- readSource sourcePointer sourceSize + case copied of + Left message -> emitResult callback context (CInt 2) message + Right source -> case Text.decodeUtf8' source of + Left _ -> emitResult callback context (CInt 1) "source is not valid UTF-8" + Right decoded -> + case compileToCorePrep (CompilerInput ".vxs" (Text.unpack decoded)) of + Left diagnostics -> emitBytesResult callback context (CInt 1) 3 (renderDiagnostics diagnostics) + Right artifacts -> + case ( encodeCore defaultCoreWireLimits (artifactOptimizedCore artifacts) + , encodeCorePrep (artifactCorePrep artifacts) + ) of + (Left issue, _) -> emitBytesResult callback context (CInt 3) 3 (show issue) + (_, Left issue) -> emitBytesResult callback context (CInt 3) 3 (show issue) + (Right core, Right corePrep) -> do + coreDelivered <- emitBytes callback context 0 (ByteString.pack core) + prepDelivered <- + if coreDelivered + then emitBytes callback context 4 (ByteString.pack corePrep) + else pure False + pure (if prepDelivered then CInt 0 else CInt 4) emitResult :: FunPtr OutputCallback -> Ptr () -> CInt -> String -> IO CInt emitResult callback context status message = diff --git a/Compiler/Headers/Visual/XSharp/Frontend.h b/Compiler/Headers/Visual/XSharp/Frontend.h index 58909c41..40a1dbfe 100644 --- a/Compiler/Headers/Visual/XSharp/Frontend.h +++ b/Compiler/Headers/Visual/XSharp/Frontend.h @@ -41,7 +41,9 @@ extern "C" /** Structured diagnostics encoded by the diagnostic wire protocol. */ VXS_FRONTEND_DIAGNOSTIC_WIRE = 2, /** Human-readable UTF-8 failure text. */ - VXS_FRONTEND_ERROR_TEXT = 3 + VXS_FRONTEND_ERROR_TEXT = 3, + /** The frontend's own CorePrep lowering; testing entries only. */ + VXS_FRONTEND_COREPREP_WIRE = 4 }; /** @brief Stable result codes returned by frontend ABI entry points. */ @@ -139,16 +141,21 @@ extern "C" const uint8_t *source, size_t source_size); - /** @brief Parse and compile a source fuzz input, delivering verified Core. + /** @brief Parse and compile a source fuzz input, delivering verified Core + * and the frontend's CorePrep lowering of that Core. * * Lexical, syntax, and semantic diagnostics are expected for arbitrary fuzz - * bytes. The native consumer receives Core only through @p output. + * bytes. On success this testing entry invokes @p output twice: first with + * `VXS_FRONTEND_CORE_WIRE`, then with `VXS_FRONTEND_COREPREP_WIRE`. Both + * buffers come from one compilation, so a harness can compare the native + * Core-to-CorePrep adapter with the frontend's lowering. Every other entry + * point delivers exactly one buffer and never emits CorePrep. * @param source Candidate source bytes; malformed UTF-8 is permitted. * @param source_size Number of bytes in @p source. - * @param output Synchronous receiver for a successful Core wire buffer. + * @param output Synchronous receiver for the Core and CorePrep buffers. * @param context Opaque state passed to @p output. - * @return `VXS_FRONTEND_DIAGNOSTICS` for rejected source, zero on Core - * output, or a negative value for an internal/ABI failure. + * @return `VXS_FRONTEND_DIAGNOSTICS` for rejected source, zero when both + * buffers were delivered, or another `vxs_frontend_status` on failure. */ int32_t vxs_frontend_fuzz_compile(const uint8_t *source, diff --git a/Documents/CORE-IR.md b/Documents/CORE-IR.md index f719d92e..472b93f4 100644 --- a/Documents/CORE-IR.md +++ b/Documents/CORE-IR.md @@ -242,7 +242,10 @@ the verified structure and materializes loop headers, exits, latches, and the distinct `for` update block. The Haskell CorePrep lowering and the native Core-to-CorePrep adapter must -build the same loop shape: +build the same program. The source fuzz harness enforces this for every +accepted source by comparing both lowerings of one compilation in a canonical +form; see [Fuzzing](FUZZING.md#coreprep-parity). In particular they must build +the same loop shape: - a `while` or `for` condition owns a dedicated header block. The statements that precede the loop stay in the incoming block, which jumps to the header diff --git a/Documents/FUZZING.md b/Documents/FUZZING.md index 81d6075f..49ca0b6e 100644 --- a/Documents/FUZZING.md +++ b/Documents/FUZZING.md @@ -38,6 +38,29 @@ subset; it is not an oracle for arbitrary Visual X# programs. Invalid source is a normal rejection, whereas internal failures and verified-model inconsistencies fail the campaign. +### CorePrep parity + +The frontend lowers its optimized Core to CorePrep, and the native pipeline +lowers the same Core again with its own adapter. Only the native result +reaches Xpp, and both lowerings are well formed, so a divergence is a +miscompile that no later verifier can see. Every source that the frontend +accepts in `source_llvm_fuzzer`, `differential_fuzzer` and `source_fuzz_smoke` +is therefore checked for parity: the testing entry of the frontend delivers +the Core and CorePrep buffers of one compilation, the harness runs the native +adapter on that Core, and the two CorePrep modules must be structurally equal. + +Both modules are compared in a canonical form. Reachable blocks are ordered +depth first from the entry, true edge before false edge, and `$`-prefixed +generated symbols are renamed in first-use order while keeping their kind. +Source symbols, types, literals, operations, operand and instruction order, +and edge roles are compared exactly; blocks unreachable from the entry are +ignored. `Compiler/Fuzzing/Tests/coreprep_parity_tests` pins what the +comparison may and may not ignore. Reintroducing the for-loop update defect +makes the smoke fail in this check before any generated code runs. + +Parity proves that the two lowerings agree, not that either is correct. The +executable oracle below remains the check against an independent expectation. + Arbitrary source and generated arithmetic use separate corpora and equal per-target time budgets. This lets source mutations reach native lowering without repeatedly creating two ORC sessions for unrelated generated code. @@ -213,6 +236,9 @@ The harnesses above are evidence about the inputs they ran, not proofs: - HPC feedback is expression-tick coverage of the frontend, without memory-safety instrumentation, and its mutator is byte- and token-level, not grammar-aware; +- CorePrep parity only covers programs the mutators and the fixed smoke + sources reach. Closures and templates have no deterministic parity source + yet, and parity cannot detect a defect both lowerings share; - prebuilt LLVM and the GHC runtime are not instrumented by any campaign. Expand these deliberately instead of equating a green workflow with completion diff --git a/Documents/TESTING.md b/Documents/TESTING.md index f35a3205..1b1b5b77 100644 --- a/Documents/TESTING.md +++ b/Documents/TESTING.md @@ -118,7 +118,7 @@ go run ./helpers/cmd/develop doctor go run ./helpers/cmd/develop test ``` -The command executes 19 Catch3 binaries, one C11 ABI contract executable and +The command executes 20 Catch3 binaries, one C11 ABI contract executable and one source/differential fuzz smoke executable on Windows 10/11, macOS Sequoia/Tahoe, Ubuntu 26.04 LTS, and Fedora 43. Bazel selects the host configuration automatically; no public test instruction diff --git a/helpers/internal/development/build.go b/helpers/internal/development/build.go index 2750aec6..8992d289 100644 --- a/helpers/internal/development/build.go +++ b/helpers/internal/development/build.go @@ -33,6 +33,7 @@ var nativeTargets = []string{ "//Compiler/Runtime/AARC/Tests:aarc_runtime_tests", "//Compiler/Runtime/AARC/Tests:aarc_c_abi_tests", "//Compiler/Fuzzing:source_fuzz_smoke", + "//Compiler/Fuzzing/Tests:coreprep_parity_tests", "//Compiler/ProjectSystem/Bridge/Tests:project_registry_tests", "//Interactive/Tests:interactive_tests", }