diff --git a/.github/workflows/engine.yml b/.github/workflows/engine.yml new file mode 100644 index 000000000..40f02ac3d --- /dev/null +++ b/.github/workflows/engine.yml @@ -0,0 +1,64 @@ +name: Engine + +on: + pull_request: + types: [opened, synchronize, reopened, ready_for_review] + paths: + - "engine/**" + - "core/**" + - ".github/workflows/engine.yml" + push: + branches: [main] + paths: + - "engine/**" + - "core/**" + - ".github/workflows/engine.yml" + +concurrency: + group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + +permissions: + contents: read + +jobs: + verify: + name: Verify the engine and the core it generates + runs-on: ubuntu-latest + timeout-minutes: 20 + + steps: + - name: Checkout code + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + + # The engine has its own toolchain: Node runs its TypeScript directly, and the generated + # Python and Go are compiled by the runner's own preinstalled toolchains. + - name: Setup Node.js + uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version-file: .nvmrc + + - name: Install the engine's dependencies + run: npm ci + working-directory: engine + + # The generated output is committed, so the three formatters decide its bytes. Prettier and + # gofmt come with the engine's install and with Go; ruff is pinned to the version in + # `engine/toolchain.lock.json`, and `verify` fails outright if any of the three is missing. + - name: Install ruff, the Python formatter + run: pipx install ruff==0.15.8 + + - name: Verify + # Compiles, generates every target in both idiom modes, regenerates and diffs for + # determinism, runs each target's linters over the output, and runs the differential + # conformance harness against the published package and across the three targets. + run: node ../engine/scripts/verify.ts . + working-directory: core + + - name: Check the committed output is what the engine emits + # `out/` is committed so it can be read in review; the verification above regenerated it, + # so a diff here means the sources changed without the output being regenerated. + run: git diff --exit-code --stat -- out + working-directory: core diff --git a/.gitignore b/.gitignore index 7a784b1d1..396521814 100644 --- a/.gitignore +++ b/.gitignore @@ -19,3 +19,6 @@ reports/* !reports/api/ reports/api/* !reports/api/brazilian-utils.api.md + +# Scratch git worktrees, created when a task is run in isolation from the checkout. +.claude/worktrees/ diff --git a/core/.gitignore b/core/.gitignore new file mode 100644 index 000000000..16dd9ea5a --- /dev/null +++ b/core/.gitignore @@ -0,0 +1,14 @@ +# The generated output of the four targets is committed on purpose: reading it is the point of +# this project, so a reviewer should see what the engine produces without having to run it. +# `npm run verify` regenerates it and fails if what is committed drifts from what the engine emits. +# +# The `--no-idioms` trees are a comparison the verification builds, not an artifact anyone reads, +# and the interpreter and build caches are not ours. +out/*-plain/ +**/__pycache__/ +out/**/.ruff_cache/ +out/**/target/ +node_modules + +# Cargo build artifacts from the benchmark harness, the same as the generated crate above. +bench/rust/target/ diff --git a/core/README.md b/core/README.md new file mode 100644 index 000000000..db56ff8f4 --- /dev/null +++ b/core/README.md @@ -0,0 +1,47 @@ +# Brazilian Utils core + +The single-source core of Brazilian Utils: each utility is written **once**, in the engine's +restricted subset, and generated for TypeScript, Python and Go. + +```sh +npm run check # types, ranges, refinements, effects +npm run build # generate every target, in both idiom modes +npm run conformance # interpreter vs the npm package vs every target +npm run verify # all of it, plus linters and a determinism check +``` + +## Layout + +``` +source/ + is-valid-cpf.ts one utility per file: one export, at the root + is-valid-cnpj.ts + format-cnpj.ts + get-holidays.ts + is-business-day.ts + get-address-info-by-cep.ts + format-currency.ts + generate-cpf.ts + generate-cnpj.ts + lib/ library code: check digits, masks, JSON, Easter, random +conformance/ + cases.ts vectors, seeded inputs and the scripted Http + run.ts the differential runner +docs/ + contracts.md what the published package does, measured + survey.md every utility of the package, by feature +out/ generated, and committed so it can be read in review +``` + +## What the core owes, and what the DX owes + +The core takes already-normalized values and returns semantic ones. Coercion stays in the +handwritten DX of each language: reading a number as a string, defaulting an options object, +turning a host `Date` into a civil date. [`docs/contracts.md`](docs/contracts.md) records that +split for every pilot, with the behavior measured from `../src` rather than assumed. + +## Status + +Nine utilities, generated for three languages, verified against the published package on every +case that can be reproduced offline. [`docs/survey.md`](docs/survey.md) counts what the rest of +the package would need. diff --git a/core/bench/README.md b/core/bench/README.md new file mode 100644 index 000000000..7b64f8234 --- /dev/null +++ b/core/bench/README.md @@ -0,0 +1,972 @@ +# Cross-language benchmark + +Answers one question per language: is the code the engine generates (`core/out//`) +faster or slower than the implementation that language's community actually ships? The ratio is +always **generated ÷ handwritten**, so below 1.0 means the generated code is faster -- the same +convention `conformance/bench.ts` already used. + +**The bar is 1.0x: equal or faster, everywhere.** Earlier revisions of this document budgeted 1.5x; +that number is gone as a target and survives below only where it names the line a row used to be +allowed to cross. `## Getting every row to 1.0x`, after the row-by-row account below, has the pass +that closed most of the gap, what moved each row and by what mechanism, the per-target inlining +budgets that pass measured rather than assumed, and the rows still over 1.0x with the structural +reason they did not come down further. + +## Running it + +```sh +node core/bench/run.mjs +``` + +from the repository root. It runs each language's harness as a subprocess -- `node` for +TypeScript, `python3` for Python, `go run` for Go -- and folds their results into one combined +table. Each harness also runs standalone if you want just one language's numbers with its own +progress output: + +```sh +cd core && node --import ./conformance/sloppy-imports.mjs ./bench/typescript.ts +python3 core/bench/python.py +cd core/bench/go && go run . +cd core/bench/rust && cargo run --release +``` + +## What this needs that isn't in this repository + +The handwritten ports are cloned read-only, outside this repository, so this benchmark cannot run +in CI -- there is nothing to attach these paths to there. If a path is missing, `run.mjs` (and +each standalone harness) prints exactly which one and the clone command that fixes it, then skips +that language and keeps going: + +They are expected in a `brazilian-utils/` directory beside this repository's checkout, which is +what the clone commands below produce. Set `BRUTILS_ROOT` to point somewhere else. + +```sh +cd .. # beside the javascript checkout +git clone https://github.com/brazilian-utils/python brazilian-utils/python +git clone https://github.com/brazilian-utils/go brazilian-utils/go +git clone https://github.com/brazilian-utils/rust brazilian-utils/rust +``` + +TypeScript needs no external clone: its handwritten side is `../src` in this same repository. + +Python's `formatCurrency` row additionally needs `num2words` installed (`pip install num2words`): +`brutils/currency.py` imports it at module scope even though `format_currency` itself never calls +it -- the same "declared but unused by this function" situation `cpf.py`/`cnpj.py` have with +`holidays`, except this one cannot be dodged by loading the file directly (see `python.py`'s +docstring), because the import is inside `currency.py` itself, not in `__init__.py`'s re-export +chain. + +The numbers in this file were measured against these commits. Pin them to reproduce the +comparison byte for byte, since the ports move independently of this repository: + +| port | commit | dated | +| --- | --- | --- | +| `brazilian-utils/python` | `330627e9d76df2c2a484ca4c6afd2ac9e20a995f` | 2026-09-11 | +| `brazilian-utils/go` | `ea155a85a012f0c5ed01a04f50a92a7edff7656c` | 2026-09-13 | +| `brazilian-utils/rust` | `a60585f7ae517adfb7389d3398da74916dd123f3` | 2026-09-14 | + +The Go harness reaches its port through a `replace` directive in `core/bench/go/go.mod`, which is +a path relative to that file rather than an absolute one, so the default layout needs no +environment at all; `run.mjs` rewrites it for the run and restores it when `BRUTILS_ROOT` is set. + +## Toolchain versions measured with + +Recorded live by each harness and printed in the combined table's "Toolchain versions" section; +as measured for the run in this report: + +- node: v22.22.2 +- python: 3.11.15 +- go: go1.24.7 linux/amd64 +- rustc: rustc 1.94.1 (e408947bf 2026-03-25) + +(Matches `engine/toolchain.lock.json`'s `measuredWith` block, which is the generator's own record +of what it was last verified against.) + +## Utilities covered + +The generated core has nine utilities. `getAddressInfoByCep` is not benchmarked anywhere -- it +makes real network calls, which a tight timed loop cannot exercise fairly -- so eight are in +scope, and the table below is what each language's harness actually compares them against. + +| utility | TypeScript | Python | Go | Rust | +| --- | --- | --- | --- | --- | +| `isValidCpf` | yes | yes | yes | yes | +| `isValidCnpj` | yes | yes | yes | yes | +| `formatCnpj` | yes | excluded | yes | excluded | +| `formatCurrency` | yes | yes | yes | yes | +| `getHolidays` | yes | excluded | excluded | excluded | +| `isBusinessDay` | yes | excluded | excluded | excluded | +| `generateCpf` | yes | yes | yes | yes | +| `generateCnpj` | yes | yes | yes | yes | +| `getAddressInfoByCep` | not benchmarked (network calls) | | | | + +No port exposes a directly comparable `formatCpf` with the same options shape as the generated +core, so it is left out for all four languages rather than force a mismatched comparison. +`formatCnpj` is excluded for Python (checksum validation, see below) and for Rust (same reason, +and it predates this pass -- `brazilian_utils::cnpj::format_cnpj` has the identical +validate-then-format contract). `getHolidays` and `isBusinessDay` are covered **only for +TypeScript**: Python's `brutils`, Go's `brazilian-utils/go` and Rust's `brazilian_utils` each +expose just `is_holiday(date, uf)` (or `IsHoliday`) -- a single-day boolean check built differently +in every port (Python wraps the third-party `holidays` package; Go and Rust hand-list state +holiday tables) -- with no function that returns a year's list and no weekend/business-day concept +at all. There is no counterpart with the same shape to compare against in any of the three, so all +six rows (two utilities × three languages) are left out; each harness's `skipped` list records the +same reason. TypeScript's own `src/get-holidays` and `src/is-business-day` are genuine counterparts +(same contract family, see `core/docs/contracts.md`), so those two rows exist for TypeScript alone. + +`generateCpf` and `generateCnpj` are covered for all four languages, but not by equality -- see +"The generator agreement rule" below. + +## Honesty notes: what each side's API actually expects + +This is the part that decides how the table is shaped, not an afterthought. + +**TypeScript.** Both the handwritten `src/*` functions and the generated +`core/out/typescript/*` functions take a value "as written" (masked or not) and do their own +normalization internally -- neither side does work the other skips. One `full-pipeline` variant +per utility is a fair comparison, and it's the one `conformance/bench.ts` already established. + +**Python.** `brutils.cpf.is_valid` / `brutils.cnpj.is_valid` do *not* strip mask characters -- +they require an already-digit (or already-alphanumeric, for CNPJ) string and return `False` on +anything else without doing any real validation work. Feeding them a masked string would not +exercise their logic at all, so every Python row is `normalized`: both sides receive the same +pre-sanitized, mask-free string. This means the generated core is doing marginally more work than +strictly necessary even here (its regex still walks the string checking for mask characters that +aren't present), which is a small, known bias in the generated core's favor being *reported*, not +hidden. + +`brutils.cnpj.format_cnpj` is excluded entirely: it calls `is_valid` first and returns `None` on a +bad checksum, while the generated `formatCnpj` never validates a checksum at all. That's not a +normalization difference you can paper over with a shared input -- it's a different contract +(validate-then-format vs. format-unconditionally). Timing them against each other would mostly +measure the checksum computation Python's port does and the generated core does not, so no number +is published for it. + +**Go -- the one with a trap in it.** `cpf.IsValid`, `cnpj.IsValid`, `cpf.Format` and `cnpj.Format` +all normalize their input themselves, via `helpers.OnlyNumbers`: + +```go +func OnlyNumbers(numbers string) string { + numericStr := regexp.MustCompilePOSIX("[0-9]+").FindAllString(numbers, -1) + return strings.Join(numericStr[:], "") +} +``` + +which **compiles a POSIX regex on every call**, then allocates a `[]string` and joins it. There is +no lower-level entry point that skips this, so it cannot be avoided by calling "the checksum part +only" -- that part is unexported. Two variants are reported for `isValidCpf` and `isValidCnpj`: + +- `full-pipeline`: both sides receive the same raw, masked input, and each does its own + normalization plus validation. This is what a real caller of either library experiences, and + it's fair because both sides are doing equivalent work. +- `normalized`: both sides receive the same pre-stripped digit-only input. This isolates most of + the checksum work from the mask-stripping work, but not all of it -- the Go port's public + functions call `OnlyNumbers` unconditionally, so even here the handwritten side still compiles a + regex and rebuilds a string on every call, a cost the generated side does not pay once its input + is already digits-only. + +`formatCnpj` is compared only as `full-pipeline`, and only the plain case: the Go port's `Format` +has no `pad`, `obfuscate` or alphanumeric (version `2`) option, so those variants aren't +comparable and are left out. CNPJ validation is compared at version `"1"` (numeric) only, for the +same reason `IsValidCnpj`'s alphanumeric path has no counterpart -- `OnlyNumbers` strips letters +before the length check, so the Go port has no alphanumeric CNPJ support at all. + +None of this is a criticism of the Go port or a suggestion to change it -- it's read-only, nothing +here modifies it, and the observation is only about what the "normalized" numbers do and don't +isolate. + +**`formatCurrency` -- no `full-pipeline` variant exists for anyone.** `core/docs/contracts.md` is +explicit: the generated core's contract always takes an already-scaled `Decimal<2>` (an integer +number of cents), never a raw float or string -- scaling a host value into that shape is DX work, +done once, outside the core. Every port's own `format_currency`/`FormatCurrency` takes a raw +float and does its own scaling internally. There is no shape in which both sides do equivalent +"raw input" work, so unlike CPF/CNPJ, there is no `full-pipeline` row to publish here in any +language -- every `formatCurrency` row is `normalized`, meaning both sides receive the value +pre-processed into the shape their own API expects, which happens to hand the generated side less +work than a caller starting from a raw float would (it skips the float-to-decimal conversion the +handwritten side always does). This is a known, disclosed bias in the generated core's favor, +reported rather than hidden, the same way the Python CPF rows disclose theirs. + +Python's `brutils.currency.format_currency` and Go's `currency.FormatCurrency` also have no +`symbol` option at all -- both always prefix `"R$ "` -- so the generated side is called with +`symbol=true` to match; there is no variant that compares the un-prefixed case for either. + +**A real disagreement, found on every non-TypeScript port: where the minus sign goes.** All three +handwritten ports checked here -- `brutils.currency.format_currency` (Python), +`currency.FormatCurrency` (Go) and `brazilian_utils::currency::format_currency` (Rust) -- format a +negative amount as `"R$ -1.234,56"`: the symbol first, then the sign, then the number. The +generated core (and the TypeScript package it is modeled on -- see `core/docs/contracts.md` and +`Intl.NumberFormat("pt-BR", { style: "currency", currency: "BRL" }).format(-1234.56)`, which +produces `"-R$ 1.234,56"`) puts the sign *before* the symbol instead. Every harness includes a +negative value in its input set specifically to surface this, and it is not papered over: each of +the three non-TypeScript harnesses reports it as a disagreement (`input=-1234.56`, +`handwritten="R$ -1.234,56"`, `generated="-R$ 1.234,56"`), and `run.mjs`'s combined table prints +all three under "DISAGREEMENTS" rather than silently excluding the value that triggers it. + +This is **not** a code-generation defect -- the generated core is doing exactly what its own +contract (traced to the published npm package's `Intl`-based output) says to do, and TypeScript's +own handwritten `src/format-currency` agrees with it on every value tested, including negative +ones. It is a genuine, independent divergence between three unrelated community reimplementations +(which all happened to agree with *each other* on sign placement) and the JS package all of this +tooling traces its contract to. Nothing here is changed to "fix" it, in either direction; it is +reported because rule 1 says a disagreement is a finding worth more than a benchmark row, and the +timed rows still run afterward on the same input set, following the same convention the CPF/CNPJ +rows already use elsewhere in this suite. + +## The generator agreement rule, and what threading randomness through a capability costs + +`generateCpf` and `generateCnpj` draw at random, so there is no fixed value for either side to +compare for equality -- a byte-for-byte match would only mean both sides used the same seed, not +that either one is correct. Every harness instead runs 500 samples through this check, before +either side is timed: + +- every value the **handwritten** port produces must validate under **both** validators: its own + port's `is_valid`/`IsValid`, and the generated core's `isValidCpf`/`isValidCnpj`; +- every value the **generated** core produces must also validate under both. + +A generator that is fast because it skips work a correct one would do -- producing a document that +fails its own port's checksum, or one the generated validator rejects -- fails this check and is +reported as a disagreement (`core/bench/typescript.ts`'s `checkGeneratorAgreement` and its +Python/Go/Rust equivalents), not silently allowed to post a number. Every language passed this +check on every sample in every run made while building this section: no generated or handwritten +value failed either validator. + +**The capability-threading difference.** The generated `generateCpf`/`generateCnpj` take a +`Capabilities` record (`env`) because the engine threads randomness as an effect, the same way it +threads HTTP and the clock -- see `core/out/typescript/capabilities.ts`'s `Capabilities` type, and +its equivalents (`python._support.Capabilities`, the Go `Capabilities` interface, +`coreout::support::Capabilities` in Rust). Every handwritten port instead calls its language's +global RNG directly (`Math.random()`, `random.randint`, `math/rand`, `rand::thread_rng()`) with no +capability object at all. This is a real, structural difference in how the two sides get +randomness, not a detail to bury -- and each harness builds its `Capabilities` value **once**, +outside the timed loop, and reuses it for all 200,000 calls, the way a real caller would build one +environment and thread it through many calls rather than rebuild it per call. Passing that +already-built value into each call is cheap; what is *not* always cheap is what `next_u32()` does +once it's called, and that is where the language-by-language story diverges -- see "What the +numbers actually showed" and the Rust section below for the measured breakdown in each language. + +## What the numbers actually showed + +Run `node core/bench/run.mjs` for current numbers; this section describes what was found while +building the harness, not a frozen result. + +- TypeScript and Python: the generated core lands close to its handwritten sibling on + `isValidCpf`/`isValidCnpj`/`formatCnpj` (TypeScript clearly faster across all three rows; Python + within a few percent either way on both rows). +- Go: the first run came back 6-9x *slower* on the generated side for `isValidCpf` and + `isValidCnpj`, on both variants, while generated `formatCnpj` -- which uses no regex -- was + about 3x faster. That split was the whole diagnosis: the Go target emitted + `regexp.MustCompile(...)` **inline in the function body**, so a large Unicode-range pattern was + recompiled from its source string on every call. `regexp` has no compilation cache, and measured + on its own the compile costs 2257.5 ms against 28.4 ms for the match over 200 000 iterations -- + **79.6x**, which is to say the benchmark was almost entirely timing regex compilation. + + The Go target now lifts every pattern into a package level `var`, the way a Go author would + have written it, and the same rows come back at **0.14x to 0.16x** -- the generated code is + roughly six times faster than the handwritten port rather than seven times slower. This is the + most valuable thing the benchmark produced, and it is the argument for benchmarking against + another language's real implementation rather than against yourself: the TypeScript benchmark + had been green for weeks and could not have found it, because JavaScript caches compiled regex + literals and Python caches compiled patterns in `re`, so only Go ever paid the cost. +- No disagreements were found on any `isValidCpf`/`isValidCnpj`/`formatCnpj`/`getHolidays`/ + `isBusinessDay` input, in any language -- see "Honesty notes" above for the one disagreement + this pass did find (`formatCurrency`'s negative-sign placement, on Python, Go and Rust, not + TypeScript), and "The generator agreement rule" for `generateCpf`/`generateCnpj`'s validity + checks, all of which passed everywhere. +- TypeScript `getHolidays` (1.66x) and `isBusinessDay` (1.73x), both mildly over budget. Neither + is caching: `src/get-holidays`'s per-year `Map` memoization looked like the obvious suspect (a + real caller benefits from it, the generated core has nothing equivalent), but defeating it -- + 200 distinct years, never repeating one -- changed the handwritten side's time by well under 1% + (186.5 ms cached vs. 188.0 ms uncached over 200,000 calls), so the cache is not what is being + measured here. Isolated with the same "time the pieces" approach the Go and Rust findings below + use: `civilDate` (`core/out/typescript/lib/civil.ts`), called once per fixed holiday to turn a + `(year, month, day)` triple into an epoch-day integer, costs 294.6 ms for 12 calls × 200,000 + iterations against a 378.8 ms whole `getHolidays` call -- about 78% of it. `civilDate` computes + the day forward (`daysFromCivil`) and then verifies the round trip by computing `yearFromDays`, + `monthFromDays` and `dayFromDays` back from it, four Howard Hinnant floor-division-heavy + functions per call; the handwritten side instead calls `new Date(year, month - 1, day)`, one + native V8 binding. That is the whole gap: portable, hand-rolled calendar math the core needs + because it has no host `Date` type (`core/docs/contracts.md`: "the core never sees a zone") costs + more than V8's own, highly optimized calendar engine. `isBusinessDay` inherits the same cost, + because it calls generated `getHolidays` on every invocation with no caching of its own. +- TypeScript `generateCpf` (54.9x) and `generateCnpj` (48.4x), the largest gap this pass found + anywhere. Isolated the same way: `randomCpfBase` (nine `env.nextU32()` draws) alone costs 4029 ms + over 200,000 calls, against 3985-4129 ms for the whole `generateCpf` call across runs -- **at + least 97%** of it, with `cpfCheckDigit` alone costing 4.9 ms, noise by comparison. + `defaultCapabilities().nextU32()` (`core/out/typescript/capabilities.ts`) is backed by + `crypto.getRandomValues(new Uint32Array(1))`; called alone, 200,000 draws cost 434.9 ms, against + 3.3 ms for 200,000 calls to `Math.random()`, the generator `src/generate-cpf` actually uses -- + **about 130x** the cost per draw. `generateCpf` makes nine of these draws per call (twelve for + `generateCnpj`), so the entire ratio is explained by one thing: the default capability's choice + of a syscall-backed CSPRNG, allocating a fresh `Uint32Array` per draw, called once per digit, + against a single native, in-process float generator called once per document on the handwritten + side. Nothing in the generated check-digit or string-building logic is at fault -- see "The + generator agreement rule" above for why the capability, not the core, is what this measures. +- Python `formatCurrency` (2.74x). `group_thousands(keep_digits(whole))` + (`core/out/python/lib/format.py` and `lib/digits.py`) alone costs 389.6 ms over 200,000 calls, + against 503.8 ms for the whole generated `format_currency` call -- about 77%. `keep_digits` is a + compiled-regex substitution and is cheap on its own; the cost is in `group_thousands`, which + turns the string into `scalars: List[int] = [ord(c) for c in whole]`, walks it calling + `trunc_mod` (a Python function call doing sign-aware modulo) once per character to decide where + a `.` goes, then rebuilds the result with `"".join(chr(p) for p in out)` -- three to six Python + function calls per character of a number that, in every value tested here, has at most six + digits. `brutils.currency.format_currency` does the whole job in one call into `Decimal`'s C + formatting machinery (`f"R$ {decimal_value:,.2f}"`) with no per-character Python loop at all. + Same shape of finding as the Rust regex story below: a small, generic, reusable primitive + (`group_thousands`, callable on a value of any length) costs more in per-character Python + function-call overhead than a native formatter built for exactly this job. +- Python `generateCpf` (3.52x) and `generateCnpj` (3.97x) -- present, but far smaller than + TypeScript's, and for a more mixed reason. `random_cpf_base` (nine `next_u32()` draws through + the rejection-sampling wrapper) costs 1946 ms over 200,000 calls against 2612 ms for the whole + `generate_cpf` call (about 75%); `cpf_check_digit` alone costs 184 ms per call, so its two calls + in `generate_cpf` account for another ~14%. Unlike TypeScript, the capability itself is not the + dominant cost: `python._support.Capabilities.next_u32` (`secrets.randbits(32)`) costs 132.8 ms + for 200,000 calls, only about 2.1x `random.randint`'s 63.2 ms baseline -- nowhere near + `crypto.getRandomValues`'s ~130x. Most of `random_cpf_base`'s cost is instead ordinary Python + function-call overhead: nine separate `random_digit(env)` → `random_below(10, env)` → + `env.next_u32()` call chains (manually timed at 1412 ms for nine raw `next_u32()` calls alone, + meaning the wrapper functions add roughly as much again on top), against + `str(randint(1, 999999998)).zfill(9)` on the handwritten side -- one draw, formatted once. + +## Rust: five lowerings, measured one at a time + +Rust was the worst target by some distance — `isValidCpf` at 2.73x, nothing under 1.0x — and the +reason turned out not to be one thing. Each of these was timed in isolation first, on a scratch +crate built against the generated one, and only then written as a candidate: + +| lowering | isolated cost, 200k calls | after | +| --- | ---: | ---: | +| `re.retain` (`keep_digits`) — a `for` loop into one `String` instead of `.filter().collect::>()` then `from_utf8().unwrap()` | 11.4–12.3 ms | 5.4–7.1 ms | +| `re_take_fixed`/`re_take_class` — test the leading byte directly, decode a `char` only above 0x7F | 6.7 ms | 4.4 ms | +| `str.padStart`, ASCII-gated — no `Vec` built just to learn a length | 22.8–24.2 ms | 10.4–10.5 ms | +| `str.codePoints` / `str.fromCodePoints`, ASCII-gated — no decode, no `char::from_u32` round trip | 5.5–5.8 / 11.6–12.1 ms | 3.5–3.6 / 8.0–8.3 ms | +| `str.fromInt` for a value proven `Int[0..9]` — one ASCII byte instead of the general integer formatter | — | — | + +The scanner fix is the one worth noticing: `re_take_fixed` and `re_take_class` are the whole of +every generated chain-pattern scanner, so making them byte-first speeds up every regex-shaped +validator this engine will ever emit, not the two rows that motivated it. + +Two findings from the same pass that are not speedups: + +- `unsafe { String::from_utf8_unchecked(..) }` for `keep_digits` measured 5.4–5.6 ms against the + safe loop's 5.4–5.5 ms — indistinguishable. The generated code stays `unsafe`-free, and now for a + measured reason rather than a stylistic one. +- The first version of the `str.fromInt` candidate returned a `raw` text fragment, which stringifies + its argument immediately. That broke `hoistConstantTables` (`backend/lower.ts`), which walks the + *structured* target AST after every candidate's `emit` has run: a weight table that had been a + module-level `const` silently became a `vec![...]` allocated on every call. Conformance did not + catch it — the answers were identical — and `cargo clippy`'s `useless_vec` did. The candidate now + builds a structured `call` node, which is the discipline the other candidates already follow, and + the hoisted constants came back. + +## Size is a result too, and it is measured the same way + +Speed is not the only thing a generated target is judged on. The npm package this engine generates +for is tree-shakeable, which +[ADR 0012](../../engine/docs/decisions/0012-generated-source-not-a-bound-binary.md) records as a +requirement rather than a preference, so for TypeScript **bytes over the wire are a benchmark +row** — and a row that is asserted rather than measured is not a row. + +`node engine/scripts/size.ts core` measures it the way a consumer's bundler would: one +single-import entry point per exported utility, bundled and minified by esbuild against +`core/out/typescript`, then compressed. Raw source bytes are the wrong number (comments, type +annotations and formatting all vanish first) and the whole tree is the wrong number too (nobody +imports all of it). + +It reports three numbers per export, because they answer different questions and this project has +already been wrong about which one matters. **Minified** is what the browser parses and the engine +holds; it is not a transfer size, but it is the only one that tracks parse and compile cost. +**Gzip** and **brotli** are both transfer sizes, and they disagree: gzip's window makes locally +repeated text almost free, so an encoding can be meaningfully *shorter raw and larger gzipped* — +which is not a hypothetical, it is what the measurement below found. Brotli weighs the same source +differently and is what most CDNs actually serve. + +Both transfer encodings are gated at zero growth, since a consumer gets whichever their CDN +negotiates. Minified is reported and not gated: trading parse cost against transfer size is an +argument to have, not a threshold to trip. `--check` compares against the committed +`core/out/typescript/SIZE.json`; `verify` runs it as its `typescript size` step, so the trade is +checked on every run rather than remembered. + +Gzipped bytes per utility, with the pass disabled, under the first (uncapped) inlining budget, and +under the budget this section settled on: + +Gzipped bytes, since that is the metric the three columns below were compared under: + +| export | no inlining | uncapped (`maxStatements: 6`) | now (8 / cap 6) | +| --- | ---: | ---: | ---: | +| `formatCnpj` | 348 | 348 | **334** | +| `formatCurrency` | 331 | 333 | **312** | +| `generateCnpj` | 642 | 1,591 | **629** | +| `generateCpf` | 623 | 1,343 | **614** | +| `getAddressInfoByCep` | 1,286 | 1,295 | **1,277** | +| `getHolidays` | 872 | 1,403 | **817** | +| `isBusinessDay` | 1,065 | 1,638 | **1,022** | +| `isValidCnpj` | 512 | 569 | **499** | +| `isValidCpf` | 342 | 440 | **333** | +| every export | 3,224 | 5,644 | **3,175** | + +Three changes got it there, and each is worth stating separately because only the first is about +inlining at all. + +- **The budget prices an inline instead of only sizing the callee** + ([ADR 0013](../../engine/docs/decisions/0013-inlining-pays-for-itself.md)). A callee spliced into + its last remaining call site costs nothing — its definition falls out of the dependency closure — + while a helper copied to nine call sites costs eight copies of itself. Alongside it, a callee that + is one `return ` is now substituted as an expression rather than through a synthetic + `Option`, and a straight-line callee is spliced without the early-return sentinel. +- **Constants the checker already proved are printed as constants.** Specialization (ADR 0004) + gives a helper called with a literal a parameter of type `Int[n..n]`, and a read of one *is* that + integer — nothing new is decided, the range on the node is the checker's own conclusion. That + turns `randomBelow`'s `4294967296 - (4294967296 % bound)` from a modulo per draw into the + constant `4294967290`, and it proves two intermediates of the Meeus Easter algorithm constant + outright for the years `easterSunday` accepts. Folding then leaves bindings nothing reads and + parameters nothing needs, which Go and Rust both refuse to compile, so `optimize.ts` removes + both — a parameter only when every call site passes something whose evaluation cannot be + noticed. This is where Go's generators moved from 0.57x/0.44x to **0.39x/0.32x**. +- **A fold over the lowered target AST** (`engine/src/backend/fold.ts`). A lowering is code + generation too: TypeScript's `date.fromYmd` expands a month into a days-in-month ladder, so once + the month is a constant the ladder is five comparisons and two branches with one possible answer. + The Core folder never saw them, because they did not exist when it ran. + +One defect surfaced on the way and is worth recording: `writeFiles` never removed generated files a +later run stopped producing, so a module that disappeared from the dependency closure — which is +exactly what inlining a helper into its only caller does — stayed on disk and was typechecked, +benchmarked and committed as though it were still output. Two stale files were in the repository. +Generated files now carry their own removal: anything under the output directory with this engine's +header that the current run did not write is deleted, and nothing else is touched. + +## A loop is smaller raw and bigger compressed — a wall, not a missed trick + +`generateCpf` and `generateCnpj` were still noisy around or over 1.0x after "Getting every row to +1.0x" (below) closed everything else. Their remaining cost traced to one thing: `randomCpfBase` +and `randomCnpjBase` build a string by drawing nine (twelve, for CNPJ) digits and converting each +one individually. A change that looked strictly better was measured against this — the generated +code is faster *and* the raw file is smaller — and it was still declined, because the one number +this target is gated on grew. That is worth writing down in full: it is a general fact about what +this engine's output looks like, not a note about two functions, and the next idea for shrinking +generated TypeScript by removing repeated call text will hit the same wall. + +**The mechanism.** `randomCpfBase`'s source (`core/source/lib/cpf.ts`) is a template literal that +calls `randomDigit()` nine times, in text, because the Core subset has no unbounded string-building +loop for an author to reach for, and because writing it as nine separate calls is what lets the +checker prove the result is exactly nine scalars long (ADR 0013 already covers why the source is +shaped this way). The generated TypeScript inherits that shape: nine — twelve, for CNPJ — +textually identical calls, `randomBelow(env).toString() + randomBelow(env).toString() + …`. Textual +identity is the whole story: the calls draw different values at runtime, but the *source text* +that makes each call is a verbatim repeat of the one before it, eight or eleven times over. + +A candidate lowering (`engine/src/backend/lower.ts`'s `operation`, gated behind a new +`TargetSpec.loopUnroll` flag set only for TypeScript) recognizes N structurally-identical pieces of +a `str.concat` chain — same call, same arguments, proven with a `sameExpr` structural comparison — +and prints one counted loop that runs the call N times instead of N copies of its text. This is +sound: the call still runs exactly N times, in the same order, so the sequence of draws the +conformance protocol fixes is unaffected — only how many times the call's *text* appears in the +file changes, which `randomCpfBase` illustrates: + +```ts +// before: nine copies of the same text +export function randomCpfBase(env: Capabilities): string { + return ( + randomBelow(env).toString() + + randomBelow(env).toString() + + // … seven more, identical … + randomBelow(env).toString() + ); +} + +// after: the same call, run nine times by a loop, with the fromCharCode-instead-of-toString+concat +// insight (below) applied once, inside the loop body, instead of once per digit +export function randomCpfBase(env: Capabilities): string { + let tmp2: number[] = new Array(9); + for (let tmp1 = 0; tmp1 < 9; tmp1++) { + tmp2[tmp1] = 48 + randomBelow(env); + } + return String.fromCharCode(...tmp2); +} +``` + +Removing eight or eleven repeats of the same ~15-character substring is a real reduction in raw +bytes — 66 fewer for `generateCpf`, 112 fewer for `generateCnpj`, minified. It is also, on its own, +close to *free* to an LZ77-family compressor: gzip's DEFLATE and brotli's own LZ77 stage both +encode a repeated substring as a short back-reference (a length and a distance) instead of its +literal bytes, so text repeated eight or eleven times costs only a few compressed bits *per +repetition*, however long the substring or however verbose the file looks on disk. The loop that +replaces it — `let t=new Array(9);for(let r=0;r<9;r++)t[r]=48+s(e);return String.fromCharCode(...t)` +— is short, but it is short, *unique*, once-occurring text: there is nothing in it for a +back-reference to point at. Trading eight-or-eleven-times-repeated text for shorter unique text is +a net loss under both codecs even though it is an unambiguous win raw, because the repeated text's +compressed cost was already close to nothing and the unique text's is not. + +**Both variants, measured.** `node engine/scripts/size.ts core` (now — see the note at the end of +this section — reporting and gating gzip *and* brotli, both at maximum quality) on the committed +baseline against the loop variant, minified / gzip / brotli, every export: + +| export | minified (base → loop) | gzip (base → loop) | brotli (base → loop) | +| --- | ---: | ---: | ---: | +| `formatCnpj` | 518 → 518 | 334 → 334 | 292 → 292 | +| `formatCurrency` | 434 → 434 | 312 → 312 | 267 → 267 | +| `generateCnpj` | 1,301 → **1,189 (−112)** | 629 → **652 (+23)** | 549 → **568 (+19)** | +| `generateCpf` | 1,284 → **1,218 (−66)** | 614 → **633 (+19)** | 532 → **551 (+19)** | +| `getAddressInfoByCep` | 2,855 → 2,855 | 1,277 → 1,277 | 1,149 → 1,149 | +| `getHolidays` | 2,379 → 2,379 | 817 → 817 | 765 → 765 | +| `isBusinessDay` | 2,922 → 2,922 | 1,022 → 1,022 | 966 → 966 | +| `isValidCnpj` | 1,481 → 1,481 | 499 → 499 | 452 → 452 | +| `isValidCpf` | 760 → 760 | 333 → 332 (−1) | 290 → 291 (+1) | +| `(all)` | 9,839 → **9,661 (−178)** | 3,175 → **3,192 (+17)** | 2,891 → **2,908 (+17)** | + +`generateCpf` and `generateCnpj` are the only rows that move, at every metric, confirmed at the +bundle level (not just by diffing the generated files — the `(all)` bundle, which puts both +functions' now-similar loop bodies in the same file where they could in principle compress against +each other, still grows by the same +17 both codecs agree on). `isValidCpf`'s ∓1 is noise, not a +third affected row: `isValidCpf` never calls `randomCpfBase` (it imports only `cpfCheckDigit`, +`cpfCheckDigit1` and `isRepeated` from the same source file), its *minified* byte count is +identical in both variants, and gzip and brotli move it in opposite directions by one byte each — +consistent with esbuild's minifier assigning a different short name to some unrelated binding +because an unrelated part of the same source file changed, not with any real content difference. + +Speed, `core/bench/typescript.ts`, generated ÷ handwritten, three runs each: + +| row | baseline (unrolled) | loop variant | +| --- | --- | --- | +| `generateCpf` | 0.92x – 1.31x (noisy, often over budget) | **0.80x – 0.91x** | +| `generateCnpj` | 1.19x – 1.31x (always over budget) | **0.68x – 0.85x** | + +The loop variant is not a marginal win — both rows go from noisy-around-or-over-1.0x to solidly +under it, every run, and it is faster than a simpler variant tried first (unrolled +`String.fromCharCode(48 + randomBelow(env), 48 + randomBelow(env), …)` in place of the +`.toString()`-and-`+` chain, same number of calls, no loop: 0.88x–0.93x / 0.90x–0.96x, +9/+9 gzip, +smaller than the loop's own +19/+23 but not zero either). + +**Why brotli, and why it agrees.** gzip's window is 32 KiB and it has no notion of "text that looks +like other text commonly seen on the web" — its preference for local redundancy could plausibly be +a gzip-specific quirk rather than a real property of what a browser downloads. Brotli has a larger +window and a static dictionary seeded from common web content, and it is what most CDNs actually +negotiate today, so it was measured specifically to test whether gzip's verdict was an artifact. +It was not: brotli (quality 11, with `BROTLI_PARAM_SIZE_HINT` set — the default quality +understates this, since it is tuned for streaming, not a static asset) grows on both rows and on +the combined bundle, by an amount close to gzip's own (`generateCpf` +19 under both codecs; +`generateCnpj` +23 gzip vs. +19 brotli — brotli's dictionary helps a little here, not enough to +flip the sign; `(all)` +17 under both). Two independent general-purpose compressors, tuned to their +respective maximums, agree on the direction. That is the basis for the decision below, not one +codec's number. + +**The decision.** This is a real, reproducible, substantial speedup — both over-budget rows move +solidly under 1.0x — declined on a stated constraint: this file's bar is 1.0x on speed, but +`engine/scripts/size.ts --check`'s bar is zero gzip *and* zero brotli growth on every export, and +this change fails both by nine to twenty-three bytes depending on the row and the codec. The engine +change (`engine/src/backend/lower.ts`'s `sameExpr`/`asAsciiDigitConversion`/`loopConcat` plus a +`TargetSpec.loopUnroll` flag, `engine/src/targets/typescript/index.ts`'s `loopUnroll: true`) was +built, measured on both codecs, and reverted rather than shipped past the gate; `core/out/typescript` +in this repository is the unrolled baseline, unchanged. If the trade above — solidly-faster +generators for 9–23 bytes of gzip/brotli growth on two exports, nothing else affected — is one +worth taking, that is a call for a human to make explicitly (with `size.ts --write`, after it +accepts the new numbers as the baseline), not one this pass makes on its own. + +`engine/scripts/size.ts` reports and gates gzip *and* brotli as of this writing (both at maximum +quality, both required not to grow); a reader checking the numbers above against a future run +should confirm the tool still measures both before comparing directly, since this section's numbers +are a snapshot of one measurement rather than something `verify` re-derives on every run. + +## Getting every row to 1.0x + +The pass this section documents took the budget from "within 1.5x" to "equal or faster, +everywhere" and moved every over-budget row it could without touching `core/source/**` (the +utilities themselves) or the checker. `node core/bench/run.mjs`'s current numbers, before and +after, language by language: + +| language | utility | before | after | fixed by | +| --- | --- | --- | --- | --- | +| typescript | `getHolidays` | 1.77x | **0.93x** | native `date.fromYmd` (below) | +| typescript | `isBusinessDay` | 1.26x | **0.66x** | inherits `getHolidays`' fix | +| typescript | `generateCpf` | 1.44x | **1.04-1.28x** | call-site inlining, 8 / cap 6 | +| typescript | `generateCnpj` | 1.24x | **1.20-1.23x** | call-site inlining, 8 / cap 6 | +| python | `formatCurrency` | 2.91x | **2.66x** | `trunc_mod`/`trunc_div` inlined; ASCII-byte `codePoints`/`fromCodePoints` | +| python | `generateCpf` | 1.86x | **1.40x** | call-site inlining, budget 12 | +| python | `generateCnpj` | 1.99x | **1.90x** | call-site inlining, budget 12 | +| go | `formatCurrency` | 1.08x | **0.92x** | ASCII-byte `re.retain`, `codePoints`/`fromCodePoints` | +| go | `generateCpf` | 0.57x | **0.39x** | constants the checker proved, and the dead parameters they left | +| go | `generateCnpj` | 0.44x | **0.32x** | the same fold | +| rust | `isValidCpf` | 2.73x | **1.44x** | ASCII-byte `re.retain`, then the pass below | +| rust | `isValidCnpj` | 1.43x | **0.96x** | the same, and now faster than the crate it compares against | +| rust | `formatCurrency` | 1.81x | **1.05x** | one-buffer `str.concatAll`; ASCII `padStart`/`codePoints` | +| rust | `generateCpf` | 2.37x | **1.68x** | one-buffer `str.concatAll` (nine-digit chain) | +| rust | `generateCnpj` | 1.89x | **1.22x** | one-buffer `str.concatAll` (twelve-digit chain) | + +Every other row was already at or under 1.0x (`isValidCpf`/`isValidCnpj`/`formatCnpj` everywhere, +Go's `generateCpf`/`generateCnpj`) and stayed there; conformance stayed 4256/4256 per target in +both idiom modes throughout, `node engine/scripts/fuzz.ts fast --seed 20260921 --count 1000` and +`full --seed 555 --count 200` both stayed clean, and `core/out` here reflects every change +regenerated and committed. + +Three shapes cover everything below: a lowering that did redundant work the target itself could +answer more cheaply (the missing candidate), a call chain whose overhead is a target property, not +a program property (per-target inlining), and a chain of allocations where one would do. + +### The missing candidate: `date.fromYmd` + +`civilDate` (`core/out/typescript/lib/civil.ts`) computed a day forward from `(year, month, day)` +and then verified the round trip by decomposing the result back through `yearFromDays`, +`monthFromDays` and `dayFromDays` -- three more Howard Hinnant floor-division-heavy functions -- +measured at 78% of `getHolidays`' call. The round trip exists to answer exactly one question, "is +`day` within the month `month` names", and a days-in-month table (28-31, February adjusted for a +leap year) answers it directly, without decomposing anything back out: once the coarse bounds hold +(year 1-9999, month 1-12, day ≥ 1), `day > daysInMonth(month, year)` is the whole check, computed +without touching the calendar math at all. The candidate that replaced the round trip +(`engine/src/targets/typescript/index.ts`) is that check plus the single forward computation -- +one Hinnant function, not four. + +**What was tried and measured slower, and kept in this account rather than quietly dropped.** The +task's own framing named `new Date(year, month - 1, day)` as the obvious native candidate, since +that is what the handwritten side calls. It was implemented first -- `Date.UTC` to construct, +reading the result back through the UTC getters to detect a silently-rolled-over date, the same +round-trip *shape* the portable calendar already used, just against V8's calendar instead of four +Hinnant functions -- and measured directly against the portable round trip in isolation (2.4 +million calls each): the portable round trip took 366-390 ms; the `Date`-based candidate took +574-608 ms, **slower**, because `Date` object construction plus three getter reads costs more than +the arithmetic it was meant to replace. The days-in-month table has no such cost (63-81 ms for the +same 2.4 million calls, a 4.5-6x improvement over the round trip) because it touches no host object +at all. The `Date`-based candidate was removed rather than kept as a fallback; every claim in this +document is a measurement of the actual candidate shipped, not of the first thing that seemed like +it should work. + +The candidate is one expression for a "cheap" (name or literal) argument and an IIFE binding each +argument once for anything else, so an argument with a real cost is never evaluated twice; a +literal call site (`civilDate(year, 1, 1)`, `getHolidays`' own shape) always takes the cheap path. + +### Per-target inlining, measured per target + +`randomDigit(env) → randomBelow(10, env) → env.nextU32()` is a three-layer call chain, run nine to +twelve times per `generateCpf`/`generateCnpj` call, in every language that has it (TypeScript, +Python, Rust). Whether that chain's overhead disappears once it is hot is a property of the target, +not of the program -- V8 elides a monomorphic closure once it is hot, CPython pays a full stack +frame per call with nothing to elide it, and rustc/LLVM inline within a crate when they judge it +worthwhile, which for a bounds-checked call inside a loop was measured not to reliably include this +shape. `engine/src/optimize/inline.ts` is a Core-to-Core pass -- target-independent by construction, +the same layer `optimize/optimize.ts`'s constant-folding and dead-code passes live in -- but it runs +once per target, from `generate` (`backend/generate.ts`), with a budget (`InlineBudget.maxStatements`) +each `Backend` declares for itself. An eligible callee (no `Fail`, no `Http`, no direct or indirect +recursion, no lambda in its body, a statement count at or under the budget) gets its parameters +bound once each to fresh names, its body renamed so two splices of the same callee never collide, +and every `return` turned into an assignment to a synthetic `Option` result -- `break`ing the loop +it is directly inside where there is one -- so the whole thing prints as ordinary statements ending +in `opt.unwrap(result)`, using intrinsics (`opt.isNone`/`opt.unwrap`) every target already lowers on +its own. Multiple rounds let an inlined callee's own call (inlining `randomDigit` exposes its call +to `randomBelow`) get a chance in a later round. + +- **Python: aggressive, budget 12.** `randomBelow`'s own body (a bounded rejection-sampling loop) is + six statements; `cpf_check_digit`/`is_repeated` are five each, and fall under the same budget + incidentally -- there is no way to size a budget that reaches `randomBelow` without also reaching + them, since they are the same size. `generateCpf` moved from 1.86x to 1.65x, `generateCnpj` from + 1.99x to 1.89x. `generate_cpf.py`'s body is now one long flattened function instead of a chain of + small ones; Python was not weighed against a bundle-size budget the way TypeScript was below, so + 12 stayed the number. Re-measured when `maxDuplicatedNodes` arrived, and deliberately left + uncapped: at a cap of 6 its generators went to 1.65x/2.11x and at 24 to 1.59x/1.87x, against + 1.40x/1.90x uncapped, so the cap costs Python exactly what it saves TypeScript. Sweeping + `maxStatements` over 6, 9, 12 and 18 moved neither generator row outside noise + (`generateCpf` 1.35-1.46x, `generateCnpj` 1.85-1.96x), so 12 stayed rather than churn a number + the measurement does not distinguish. +- **TypeScript: `maxStatements: 8, rounds: 4, maxDuplicatedNodes: 6`, and the cap is the point.** + The first version of this budget was `maxStatements: 6` with no cap, and the note here recorded + its cost as "+20% of the generated tree" -- raw source bytes, which is the wrong number twice + over: comments, type annotations and formatting all vanish before a browser sees any of it, and + nobody imports the whole tree. `engine/scripts/size.ts` measures the right one, bundling a + single-import entry point per utility with esbuild and gzipping it. Measured that way the + uncapped budget cost **+75% across every export and +148% on `generateCnpj` alone**, for two rows + it moved by about a tenth each -- inside the run-to-run noise of a generator whose own retry loop + is random. `generate-cpf.ts` was 45 lines before the pass and 408 after: nine unrolled copies of + a rejection-sampling loop, one per digit. + + `maxDuplicatedNodes` fixes that by pricing an inline rather than only sizing the callee, and + [ADR 0013](../../engine/docs/decisions/0013-inlining-pays-for-itself.md) has the mechanism. Both + numbers were then swept against the measurement -- the cap at 0, 6, 12, 24 and 48, the statement + budget at 6, 8, 10, 12, 16, 24 and 48 -- and 8/6 came out smallest on every single export. The + result is that inlining is no longer a trade for this target at all: **every export is smaller + than with the pass disabled** (3,175 bytes gzipped across all nine, against 3,224 with no + inlining and 5,644 under the uncapped budget), and `generateCpf` still lands at 1.04-1.28x and + `generateCnpj` at 1.20-1.23x. `verify`'s `typescript size` step holds it there against the + committed `core/out/typescript/SIZE.json`. +- **Rust: none, deliberately, and the reason is itself a finding.** A first attempt at budget 6 + measured code that was *worse*, not better. This pass has no notion of a Rust borrow + ([ADR 0010](../../engine/docs/decisions/0010-rust-parameters-borrow-where-sound.md)) -- it binds + every inlined parameter as an owned local (`let name = argExpr;`), so a splice of a callee whose + parameter ADR 0010 had proven could borrow its argument printed a `.to_owned()` at that binding + instead: one full string clone per digit drawn, in `is_valid_cpf` and everywhere else the same + shape appeared. That is exactly the allocation ADR 0010 exists to remove, and it cost more than + the call overhead this pass would have saved, so `inlineBudget` is deliberately absent from + `RUST_BACKEND` (`engine/src/targets/rust/index.ts` carries the same account at the call site). A + smaller ask -- an `#[inline]` attribute hint on small non-exported functions, asking rustc's own + inliner to look rather than manually splicing anything -- was tried next and measured to change + nothing: `isValidCpf` came back at 22-25 ms with the hint present or absent across six runs, + indistinguishable from run-to-run noise, meaning rustc was already making the same inlining + decision either way. That was reverted too, rather than kept as a change with no measured effect + behind it. Closing this gap for Rust means teaching the pass to reconstruct ADR 0010's borrow + analysis for an inlined splice, not extending the budget -- future work, not this pass. +- **Go: not attempted.** Every Go row using this call shape (`generateCpf`, `generateCnpj`) was + already at or under 1.0x before this pass; `formatCurrency`, Go's one over-budget row, was closed + by the allocation fix below without touching call structure at all. + +### Allocation: one buffer or one pass, not a chain + +**Rust's string assembly.** A `+` chain (`a + b + c`) lowers as nested `str.concat` calls, +`concat2(concat2(a, b), c)`, and each `concat2` allocates a fresh `String` and copies everything to +its left into it -- `format_currency`'s `prefix`/`sign`/`body` assembly and every `concat2` in +`random_cpf_base`'s nine-digit chain were exactly this shape. `backend/lower.ts`'s `operation` now +flattens a chain of three or more pieces into its leaves before lowering (`(a + b) + c` is +`str.concat(str.concat(a, b), c)` in Core; the leaves are `[a, b, c]`, left to right, evaluated in +that order either way) and hands them to a new op, `str.concatAll`, when a target declares a +candidate for it -- Rust only, today; every target without one falls straight through to the +unchanged pairwise `str.concat` path, so nothing changed for TypeScript, Python or Go. Rust's +`str.concatAll` sizes one buffer once, from every piece's own length summed, and pushes each piece +into it once. Two correctness details worth naming because a first version of this got both wrong: +a piece whose length has to be computed (anything that is not already a name or a literal, most +often a call like `group_thousands(keep_digits(&whole))`) is bound to a local first, so it is +computed once, not twice -- once for sizing the buffer, once for pushing it, which the first version +of this candidate did not do, and which a differential run caught (`generateCpf`/`generateCnpj` +producing a well-formed but wrong document -- a different draw count changes which digits come out, +and every digit is still individually valid, so nothing but the reference interpreter's own answer +catches it); and a single-character literal piece prints `.push('x')`, not `.push_str("x")`, which +`clippy::single_char_add_str` (default warn) asks for. `generateCpf` moved from 2.37x to 1.68x and +`generateCnpj` from 1.89x to 1.22x; `formatCurrency` from 1.81x to 1.74x (smaller, because its own +remaining cost is elsewhere -- see the Rust section below). + +**`re.retain` and `codePoints`/`fromCodePoints`, in three languages.** `keep_digits`/ +`keep_alphanumeric` (`re.retain`) walked the input as decoded Unicode scalars -- +`.chars().filter(...)` in Rust, `strings.Map` in Go, `[ord(c) for c in whole]` in Python -- before +ever testing one against a range. Every class this project retains is ASCII (digits, upper and +lower case letters), and an ASCII byte needs no decoding to be tested: a multi-byte scalar's bytes +are all ≥ 0x80, so each one fails an ASCII range test on its own, exactly as testing the decoded +scalar would have failed it -- a byte-wise scan is not an approximation of the scalar-wise one, it +is the same predicate for less work, sound for any input, ASCII or not. Rust +(`String::from_utf8(value.bytes().filter(...).collect::>())`) and Go (a `[]byte` scan +building a `[]byte` result) both got this candidate, gated on every retained range being ≤ 127 (a +class outside that range, none exist in this project today, still falls back to the scalar-wise +pass). `str.codePoints`/`str.fromCodePoints` (`group_thousands`' `out: IntRange<0,127>[]`, proven +ASCII by its own declared type) got the matching treatment in Python (`.encode("ascii")` / +`bytes(...).decode("ascii")`) and Go (`[]byte(value)` / one `string()` from a `[]byte`), gated the +same way -- on the *element* type's range for `fromCodePoints`, since that op takes a list of code +points, not a string. `trunc_mod`/`trunc_div` (Python's truncated remainder/division, needed +because `%`/`//` floor in Python but the Core's contract rounds toward zero) were Python-level +functions called on every `%`/`//` whose operands were not provably non-negative -- a real cost in +CPython, where a function call is a full frame, not the arithmetic itself -- so both now print +their own formula inline as a single expression (a tuple evaluates both operands into fixed names +once, unconditionally, before the ternary that uses them picks a branch, so an operand that is +itself a call is still evaluated exactly once). Together these moved Python's `formatCurrency` from +2.91x to 2.66x and Go's from 1.08x to 0.93x -- Go's crossed 1.0x; Python's did not, and the +"Remaining over 1.0x" note below says why. + +### Remaining over 1.0x, and why + +**Python: `formatCurrency` (2.66x), `generateCpf` (1.65x), `generateCnpj` (1.89x).** +`group_thousands`' own loop -- `for index in range(0, len(scalars)): ... out.append(...)` -- is +still a Python-level loop with a method call (`list.append`) every iteration; removing `trunc_mod`'s +call and switching to bytes removed real overhead (2.91x → 2.66x) but did not remove the loop +itself, which is where most of what is left lives. Nothing in `engine/src/targets/python/**` can +turn an author's explicit per-character loop into a single C-level format call without either +recognizing that specific algorithm's shape (which this project treats as out of bounds -- a +candidate is for an *operation*, never a pattern-match on one author's algorithm) or rewriting +`group_thousands` itself, which is `core/source/lib/format.ts`, out of scope for this pass. The +generators' remaining gap is `random_cpf_base`/`random_cnpj_base`'s draw chain: even with the whole +`random_digit → random_below → next_u32` chain inlined (above), each draw is still a rejection- +sampling loop plus an `Optional[int]` bookkeeping the inlining introduced, run nine to twelve times, +against one `randint()` call on the handwritten side. Closing the remaining gap would mean either a +larger inlining budget (tried at the whole-chain size already; `cpf_check_digit`/`is_repeated` are +already incidentally included) or a structural change to how the draw itself works, both outside +what a lowering-only pass can do. + +**Rust: `isValidCpf` (2.57x), `formatCurrency` (1.74x), `generateCpf` (1.68x).** `isValidCpf`'s +`keep_digits` no longer decodes UTF-8 (above), and `cpf_check_digit`'s loop no longer clones (ADR +0010, already landed before this pass); `#[inline]` on `digit_at` was tried and measured to change +nothing (above), meaning the remaining ~2.5x is not a missed-inlining question. What is left is +spread across `re_match_3` (the CPF pattern's own scanner), the mask-character `trim_matches` every +`isValidCpf`/`isValidCnpj` call pays before `keep_digits` ever runs, and `keep_digits`' own +`String::from_utf8` allocation -- none individually dominant the way `.chars()` or the cloning loop +once were, which is itself the finding: the easy, single-cause wins are gone, and what remains is +distributed across several small, already-minimal operations, the last of which is a systems +language with an 11-character allocation and a bounds-checked scan against a handwritten crate +doing the same in roughly a third of the calls. `formatCurrency`'s remaining cost is +`group_thousands`/`keep_digits` themselves (now byte-based, but still O(n) passes CPython -- no, +Rust -- still pays for) plus `pad_start`'s own allocation, none of them a chain anymore. +`generateCpf`'s remaining 1.68x is the same rejection-sampling-loop-times-nine-to-twelve shape +Python's is, minus Python's per-call frame cost -- `next_u32()` itself is cheap in Rust, so what is +left is the loop and the `Option`/`Vec` bookkeeping around each draw, not a single hot line. +None of these three rows moved to a worse state than before this pass; all three are measurably +better than they were, and none reached 1.0x, which is reported here rather than left unexplained. + +## Rust + +`core/bench/rust/` compares the generated crate against `brazilian_utils`. Both sides take a +digits-only value -- `brazilian_utils::cpf::validate` rejects anything else before doing any work +-- so the Rust rows are `normalized` only, the same shape as the Python rows rather than Go's two +variants. `cnpj::format_cnpj` is left out for the same reason Python's is: it validates the +checksum and answers `None` for a bad one, while the generated `format_cnpj` never validates. + +The generated crate **was correct and slow**: no disagreement on any input, and 50x / 24x the time +of the handwritten crate. The cost was not spread around, it was one thing. Timed against the +generated crate directly, before the fix: + +| | 200 000 iterations | +| --- | --- | +| `support::re_test` alone | 266.9 ms | +| `keep_digits` alone | 11.7 ms | +| one `String` clone (baseline) | 3.4 ms | +| `is_valid_cpf` whole | 339.5 ms | + +`std` has no regex, so the Rust target generated its own matcher: a `ReNode` tree walked by an NFA +simulation that returned a fresh `Vec` of reachable positions per node per position, sorting +and deduplicating each one. That interpretation was 79% of the call. + +**The fix.** The engine knows every pattern in a project before it generates any Rust, so there is +no reason to interpret a pattern tree at call time at all. Each pattern now compiles to a dedicated +scanner -- straight-line Rust for that exact pattern, no allocation, one forward pass over `&str` +-- when it is a chain of character-class runs with no alternation, which is every pattern +`core/source` uses; a pattern that needs more than that (alternation, a repeated group) falls back +to a backtracking matcher over `&str` slices, also allocation-free, ported directly from the +engine's own reference regex matcher. `engine/docs/targets/rust.md` has the full account, and +`engine/src/targets/rust/index.ts`'s "Regex" section and `LOWERING.md`'s `re.test` row name the +rule that decides which pattern gets which. Re-measured the same way: + +| | 200 000 iterations | +| --- | --- | +| `support::re_match_3` alone (the CPF pattern's scanner) | 4.4 ms | +| `keep_digits` alone | 11.2 ms | +| one `String` clone (baseline) | 2.6 ms | +| `cpf_check_digit(_, 9)` alone (includes its own `keep_digits`) | 44.9 ms | +| `is_valid_cpf` whole | 78.0 ms | + +The matcher went from 79% of the call to noise: `support::re_match_3` costs about what walking an +11-character string once should. `isValidCpf` and `isValidCnpj` came down from **50x/24x to +10.8x/3.6x** the handwritten crate -- a 4.6x and 6.7x speedup respectively, and conformance stayed +4256/4256 in both idiom modes throughout, because nothing about the fix changes what a pattern +accepts (`node --test` also still passes `engine/tests/regex.spec.ts`, and every pattern +`core/source` uses classifies as the scanner shape, so the fallback matcher is unexercised code on +this project, not a silent second answer to any of these four patterns). + +**What was left after the regex fix, and why it was not "the same thing, still there".** The +bottleneck was not the regex at all: it was `cpf_check_digit`'s loop, `sum += digit_at(cpf.to_owned(), +index)` run 9 or 10 times per call. Every value was owned wherever it was bound ([ADR +0009](../../engine/docs/decisions/0009-rust-values-are-owned.md)), so each call to the +`digit_at(value: String, ...)` helper needed its own owned copy of the 11-digit string -- one +`to_owned()` per loop iteration, not one per top-level call, because the loop called a +String-taking helper instead of indexing bytes directly. `cpf_check_digit(_, 9)` alone, at 44.9 ms, +accounted for most of the remaining gap; the other `cpf_check_digit(_, 10)` call, `is_repeated`, and +two more `digit_at` calls in `is_valid_cpf` itself made up the rest. `cnpj_check_digit` never had +this cost -- its source indexes `cnpj.as_bytes()[index]` directly rather than calling a +String-taking helper in its loop -- which is exactly why `isValidCnpj`'s gap (3.6x) was so much +smaller than `isValidCpf`'s (10.8x) despite CNPJ validating more digits: the difference was which +loop-body shape `core/source` happened to use, not the pattern each one matched. Against Go, on +this same benchmark, generated Rust was already behind on both rows (`isValidCpf` 79.7 ms vs. Go's +39.4 ms; `isValidCnpj` 52.3 ms vs. Go's 49.2 ms) -- a systems language with an ownership model +losing to one with a garbage collector, because the ownership model was not being used for +anything. + +**The fix.** [ADR 0010](../../engine/docs/decisions/0010-rust-parameters-borrow-where-sound.md) +revisits ADR 0009's parameter-ownership choice with what ADR 0009 itself was missing: a pre-pass +over the *whole* Core program (`engine/src/analysis/borrows.ts`), run once before lowering starts, +deciding which `String`/`Enum`/`List` parameters are only ever read and may print as `&str`/`&[T]` +instead of being cloned at every call site that hands them one. `digit_at` and `cpf_check_digit` +both only read their string argument, so both now borrow it; `cpf_check_digit`'s loop is +`digit_at(cpf, index)` -- a bare `&str` copy, not a clone -- at every iteration, and +`is_valid_cpf`'s own `cpf` parameter borrows too, since it only ever forwards `cpf` into +`keep_digits` and the trim. `cnpj_check_digit(cnpj: &str, weights: &[i64])` borrows both of its +parameters the same way, including the hoisted weight table (`LIB_CNPJ_TABLE1: &[i64]`, already a +reference -- passed bare, not `&`-wrapped again). Re-measured the same way as the table above: + +| | 200 000 iterations | +| --- | --- | +| `support::re_match_3` alone (the CPF pattern's scanner, unchanged by this fix) | 4.4 ms | +| `keep_digits` alone (the new largest single cost) | ~9.3 ms | +| `is_repeated` alone | ~1.4 ms | +| `cpf_check_digit(_, 9)` + `cpf_check_digit(_, 10)` together (was 44.9 ms for one of the two) | ~2.2 ms | +| `is_valid_cpf` whole | 16.8 ms | + +**`isValidCpf` and `isValidCnpj` came down from 10.8x/3.6x to 2.22x/1.43x** the handwritten crate +(`node core/bench/run.mjs`: `isValidCpf` 7.6 ms handwritten vs. 16.8 ms generated; `isValidCnpj` +15.3 ms vs. 21.8 ms), and **both now beat generated Go** on the same benchmark (Go: `isValidCpf` +52.7-59.5 ms normalized/full-pipeline, `isValidCnpj` 53.1-66.6 ms -- generated Rust is now 3-4x +*faster* than generated Go on both rows it used to lose). Conformance stayed 4256/4256 in both +idiom modes throughout (nothing about a borrowed parameter's declared type changes what it +accepts -- `toOwned` converts a borrow to an owned value exactly the way it already converted an +owned value to another one), `cargo clippy --offline -- -D warnings` and `cargo fmt --check` are +clean on the regenerated crate, and `Cargo.lock` still lists exactly one package. + +**What is left.** With the loop's clones gone, the largest remaining cost in `is_valid_cpf` is +`keep_digits` itself: `.chars().filter(...).collect::()` has to build the fresh, owned +11-digit string the function returns, and `.chars()` decodes UTF-8 scalar by scalar rather than +scanning bytes directly. At ~9.3 ms of `is_valid_cpf`'s 16.8 ms, it is now close to 60% of the +call, next to `re.test`'s ~4.4 ms (unchanged -- this fix does not touch regex) and well under 2 ms +for `is_repeated` plus the now-cheap `cpf_check_digit` calls combined. `isValidCnpj`'s 1.43x is now +the *better*-behaved of the two rows: its remaining cost is almost entirely `keep_digits` on a +14-character input plus the two now-cheap check-digit calls, with no loop-shaped cost left to +distinguish it from `isValidCpf` the way the account above had it. Closing the `keep_digits` gap +further -- a byte-oriented ASCII filter instead of a `char` iterator, say -- is a different fix +than this one and is not part of it. + +### formatCurrency, generateCpf, generateCnpj + +`formatCurrency` is `normalized` for the reason every language's currency row is (see "Honesty +notes" above): the generated core always takes a pre-scaled `Decimal<2>` `i64`, so there is no +`full-pipeline` shape to compare. `getHolidays` and `isBusinessDay` are excluded for Rust for the +same reason as Python and Go: `brazilian_utils::date_utils` exposes only +`is_holiday(NaiveDate, Option<&str>) -> Option`, a single-day check with no year-list form +and no business-day concept, so there is no comparable counterpart. `generateCpf`/`generateCnpj` +are covered by the validity rule in "The generator agreement rule" above, not equality; the +`Capabilities` implementation the harness builds for them (`BenchCapabilities`) is a small +SplitMix64 seeded from the system clock -- a real, changing-per-run, non-cryptographic generator, +the same kind of thing `rand::thread_rng()` is, which is what `brazilian_utils::cpf::generate` and +`cnpj::generate` already use, so (unlike TypeScript's `crypto.getRandomValues`) the RNG's own +implementation cost is not a large factor here. + +These three rows are not the ones ADR 0010 targeted (only `isValidCpf`/`isValidCnpj` were over +budget against Go; these three were never compared against Go at all, only against the handwritten +Rust crate, and stayed over *that* budget before and after the fix), but `generate_cpf` and +`generate_cnpj` both call `cpf_check_digit`/`cnpj_check_digit` in their own check-digit step, so +they picked up part of the same win. `node core/bench/run.mjs`: `formatCurrency` 2.00x (37.6 ms +handwritten vs. 75.4 ms generated, essentially unchanged by the borrowing fix -- see why below), +`generateCpf` 2.78x, down from 4.34x before ADR 0010 (50.6 ms vs. 140.8 ms), `generateCnpj` 2.03x, +down from 2.28x (85.1 ms vs. 173.2 ms). Timed against the generated crate directly, the same way +the regex fix above was attributed: + +| | 200 000 iterations | +| --- | --- | +| `group_thousands("1234")` alone | 20.8 ms | +| `keep_digits("1234")` alone | 4.9 ms | +| `format_currency(123456, true)` whole | 76.8 ms | +| `cpf_check_digit("12345678900", 9)` alone | 0.1 ms | +| `random_cpf_base(&env)` alone (9x `next_u32` through the rejection-sampling wrapper) | 108.9 ms | +| `generate_cpf(&env)` whole | 139.5 ms | +| `random_cnpj_base(&env)` alone (12x `next_u32` through the wrapper) | 141.6 ms | +| `generate_cnpj(&env)` whole | 182.8 ms | + +`group_thousands` and `keep_digits` together are only ~33% of `format_currency`'s call (25.7 ms of +76.8 ms); the rest is allocation, not computation -- `pad_start` and each `crate::support::concat2` +in the assembly chain builds a fresh owned `String`, and every intermediate result +(`digits`, `whole`, `cents`, `body`, `prefix`) is one. This is the same shape of cost the CPF/CNPJ +section above documents, but ADR 0010's fix does not reach it: `format_currency`'s own signature +takes `value: i64` and returns an owned `String` -- there is no `String`/`List` parameter here for +the borrow pre-pass to find borrowable in the first place, and the function's job (building a new +formatted string) inherently allocates regardless of ownership. + +For `generateCpf`/`generateCnpj`, `cpf_check_digit` is no longer a factor at all (0.1 ms for one +call, the same near-zero cost the CPF/CNPJ section above measures for the pair together): the +entire remaining cost is `random_cpf_base`/`random_cnpj_base` -- about 109 ms of 140 ms and 142 ms +of 183 ms respectively. Both call `random_digit`/`random_below` once per digit (9 times for CPF, 12 +for CNPJ), and each of those returns an owned `String` (`.to_string()`) that then gets threaded +through a `concat2` chain to build the base -- the same "many small owned allocations, one per +digit, chained together" pattern TypeScript's capability section above describes for a different +reason (there, the RNG call itself is the expensive part; here, `next_u32()` is cheap and the +wrapping and String-building around each draw is what adds up). None of `random_digit`/ +`random_below`'s own parameters are `String`/`List` either (they take and return `i64`/`String` +built fresh each call, nothing to borrow), so this cost sits outside what ADR 0010 changes, the +same way `format_currency`'s does. Closing either gap, if it is worth closing, is separate work +from parameter borrowing. + +**Update -- the `concat2` chain, closed.** The gap this subsection describes (`random_cpf_base`/ +`random_cnpj_base` threading each digit's owned `String` through a `concat2` chain, and +`format_currency`'s own `prefix`/`sign`/`body` assembly doing the same) is exactly what +`str.concatAll`'s one-buffer assembly, in "Getting every row to 1.0x" above, was built for. +`node core/bench/run.mjs`'s current numbers: `generateCpf` 1.68x (was 2.37x, closer to this +subsection's 2.78x before that), `generateCnpj` 1.22x (was 1.89x), `formatCurrency` 1.74x (was +1.81x -- smaller, because a `re.retain` fix also in that section, not the buffer, was +`format_currency`'s bigger factor). `isValidCpf`/`isValidCnpj` moved the same pass from 2.73x/1.43x +to 2.57x/1.35x, for the `keep_digits` byte-scan reason that section names, not this one. Full +before/after table and per-row attribution: "Getting every row to 1.0x" above. diff --git a/core/bench/go/bench_main.go b/core/bench/go/bench_main.go new file mode 100644 index 000000000..d8c048b03 --- /dev/null +++ b/core/bench/go/bench_main.go @@ -0,0 +1,343 @@ +// Benchmarks the generated Go core against brazilian-utils/go, the handwritten package it +// replaces. Run with `go run .` from this directory (see core/bench/README.md). +// +// Convention shared by every language harness in this directory: 20,000-iteration warm-up, +// 200,000 timed iterations, same process, same inputs, 1.5x budget, ratio reported as +// generated / handwritten. +// +// Fairness note on the two variants below. Unlike the Python and (eventually) Rust ports, the Go +// port's public API (cpf.IsValid, cnpj.IsValid, cpf.Format, cnpj.Format) always normalizes its +// input itself, via helpers.OnlyNumbers -- there is no lower-level entry point that skips it. So: +// +// - "full-pipeline": both sides receive the same raw, masked input, and each does its own +// normalization plus validation/formatting. This is the comparison a real caller of either +// library experiences, and it is fair because both sides are doing equivalent work. +// - "normalized": both sides receive the same pre-stripped digit-only input. This isolates most +// of the checksum/formatting work from the mask-stripping work, but not all of it: the Go +// port's public functions call helpers.OnlyNumbers unconditionally, so even here the +// handwritten side still compiles a POSIX regex and re-joins a []string on every call, a cost +// the generated side does not pay once its input is already digits-only. This is reported as +// an observation about what the "normalized" numbers contain, not a criticism of the port -- +// it is read-only and nothing here suggests changing it. +// +// CNPJ is compared at version "1" (numeric) only: brazilian-utils/go has no alphanumeric CNPJ +// support at all (OnlyNumbers strips letters before the length check), so there is no fair way to +// exercise the version "2" path against it. +// +// formatCurrency has no full-pipeline shape at all: the generated core's contract +// (core/docs/contracts.md) always takes an already-scaled Decimal<2> int, never a raw float -- +// scaling is DX work, done once outside the core, on both sides equally here. It is therefore +// "normalized" for the same reason the CPF/CNPJ normalized rows are: pre-processed input on both +// sides. currency.FormatCurrency also has no symbol option (it always prefixes "R$"), so the +// generated side is called with symbol=true to match. +// +// getHolidays and isBusinessDay are not covered for Go: brazilian-utils/go's date package exposes +// only date.IsHoliday(time.Time, uf) -- a single-day boolean check, not a function returning a +// year's list, and it has no weekend/business-day concept at all. There is no fair counterpart, so +// both rows are left out; see core/bench/README.md. +// +// generateCpf and generateCnpj draw at random, so there is no fixed value to compare for equality; +// see the README for the "does every value validate" rule used instead. NextU32 below is backed by +// math/rand, the same non-cryptographic generator brazilian-utils/go's own cpf.Generate and +// cnpj.Generate use, so the RNG choice itself is not what a lopsided ratio would be measuring here. +package main + +import ( + "encoding/json" + "fmt" + "math/rand" + "os" + "runtime" + "time" + + core "coreout" + hcnpj "github.com/brazilian-utils/go/cnpj" + hcpf "github.com/brazilian-utils/go/cpf" + hcurrency "github.com/brazilian-utils/go/currency" +) + +// benchCapabilities is the "real" (non-fixture) environment the generated generateCpf/generateCnpj +// need: only NextU32 is ever called by them, but the Capabilities interface requires all four +// methods, so the other three are stubs that panic if ever reached. +type benchCapabilities struct{} + +func (benchCapabilities) Request(request core.HttpRequest) *core.HttpResponse { + panic("not used by generateCpf/generateCnpj") +} +func (benchCapabilities) Now() int { panic("not used by generateCpf/generateCnpj") } +func (benchCapabilities) Sleep(milliseconds int) { panic("not used by generateCpf/generateCnpj") } +func (benchCapabilities) NextU32() int { return int(rand.Uint32()) } + +const budget = 1.5 +const warmup = 20_000 +const iterations = 200_000 + +// Raw, masked test vectors -- the numeric-only subset of the ones typescript.ts and python.py +// use, since the Go port has no alphanumeric CNPJ support (see package doc above). +var rawCPFs = []string{"123.456.789-09", "12345678909", "00000000000", "529.982.247-25"} +var rawCNPJs = []string{"12.345.678/0001-95", "12345678000195", "00000000000000"} + +func onlyDigits(s string) string { + out := make([]byte, 0, len(s)) + for i := 0; i < len(s); i++ { + if s[i] >= '0' && s[i] <= '9' { + out = append(out, s[i]) + } + } + return string(out) +} + +func mapStrings(in []string, f func(string) string) []string { + out := make([]string, len(in)) + for i, v := range in { + out[i] = f(v) + } + return out +} + +var normalizedCPFs = mapStrings(rawCPFs, onlyDigits) +var normalizedCNPJs = mapStrings(rawCNPJs, onlyDigits) + +// Raw floats, the shape currency.FormatCurrency's own callers use. Includes a negative value on +// purpose -- see the README's "formatCurrency" honesty note for what that turns up. +var currencyValues = []float64{0, 1234.56, -1234.56, 0.5, 999999.99, 10} + +func toCents(value float64) int { + if value < 0 { + return -int(-value*100 + 0.5) + } + return int(value*100 + 0.5) +} + +type row struct { + Utility string `json:"utility"` + Variant string `json:"variant"` + HandwrittenMs float64 `json:"handwrittenMs"` + GeneratedMs float64 `json:"generatedMs"` + Iterations int `json:"iterations"` +} + +type disagreement struct { + Utility string `json:"utility"` + Variant string `json:"variant"` + Input interface{} `json:"input"` + Handwritten interface{} `json:"handwritten"` + Generated interface{} `json:"generated"` +} + +var rows = []row{} +var disagreements = []disagreement{} + +func checkAgreementBool(utility, variant string, inputs []string, handwritten, generated func(string) bool) { + for _, input := range inputs { + a := handwritten(input) + b := generated(input) + if a != b { + disagreements = append(disagreements, disagreement{utility, variant, input, a, b}) + } + } +} + +func checkAgreementString(utility, variant string, inputs []string, handwritten, generated func(string) string) { + for _, input := range inputs { + a := handwritten(input) + b := generated(input) + if a != b { + disagreements = append(disagreements, disagreement{utility, variant, input, a, b}) + } + } +} + +func measure(run func()) float64 { + for i := 0; i < warmup; i++ { + run() + } + started := time.Now() + for i := 0; i < iterations; i++ { + run() + } + return float64(time.Since(started)) / float64(time.Millisecond) +} + +func compareBool(utility, variant string, inputs []string, handwritten, generated func(string) bool) { + checkAgreementBool(utility, variant, inputs, handwritten, generated) + + fmt.Printf("%s (%s)\n", utility, variant) + cursor := 0 + handwrittenMs := measure(func() { + handwritten(inputs[cursor%len(inputs)]) + cursor++ + }) + fmt.Printf(" handwritten %.1f ms\n", handwrittenMs) + + cursor = 0 + generatedMs := measure(func() { + generated(inputs[cursor%len(inputs)]) + cursor++ + }) + fmt.Printf(" generated %.1f ms\n", generatedMs) + + rows = append(rows, row{utility, variant, handwrittenMs, generatedMs, iterations}) +} + +func compareString(utility, variant string, inputs []string, handwritten, generated func(string) string) { + checkAgreementString(utility, variant, inputs, handwritten, generated) + + fmt.Printf("%s (%s)\n", utility, variant) + cursor := 0 + handwrittenMs := measure(func() { + handwritten(inputs[cursor%len(inputs)]) + cursor++ + }) + fmt.Printf(" handwritten %.1f ms\n", handwrittenMs) + + cursor = 0 + generatedMs := measure(func() { + generated(inputs[cursor%len(inputs)]) + cursor++ + }) + fmt.Printf(" generated %.1f ms\n", generatedMs) + + rows = append(rows, row{utility, variant, handwrittenMs, generatedMs, iterations}) +} + +func compareCurrency(utility, variant string, values []float64, handwritten func(float64) string, generated func(float64) string) { + for _, v := range values { + a := handwritten(v) + b := generated(v) + if a != b { + disagreements = append(disagreements, disagreement{utility, variant, v, a, b}) + } + } + + fmt.Printf("%s (%s)\n", utility, variant) + cursor := 0 + handwrittenMs := measure(func() { + handwritten(values[cursor%len(values)]) + cursor++ + }) + fmt.Printf(" handwritten %.1f ms\n", handwrittenMs) + + cursor = 0 + generatedMs := measure(func() { + generated(values[cursor%len(values)]) + cursor++ + }) + fmt.Printf(" generated %.1f ms\n", generatedMs) + + rows = append(rows, row{utility, variant, handwrittenMs, generatedMs, iterations}) +} + +const generateSamples = 500 + +// checkGeneratorAgreement implements the README's generator agreement rule: generateCpf and +// generateCnpj draw at random, so there is nothing to compare for equality. Instead, every value +// either side produces must validate under BOTH validators -- its own port's and the generated +// core's -- before either side is timed. +func checkGeneratorAgreement(utility, variant string, handwrittenGenerate func() string, handwrittenIsValid func(string) bool, generatedGenerate func() string, generatedIsValid func(string) bool) { + for i := 0; i < generateSamples; i++ { + fromHandwritten := handwrittenGenerate() + if !handwrittenIsValid(fromHandwritten) { + disagreements = append(disagreements, disagreement{utility, variant, fromHandwritten, "rejected by its own port's validator", "n/a"}) + } + if !generatedIsValid(fromHandwritten) { + disagreements = append(disagreements, disagreement{utility, variant, fromHandwritten, "valid (own validator)", "rejected by the generated core's validator"}) + } + + fromGenerated := generatedGenerate() + if !generatedIsValid(fromGenerated) { + disagreements = append(disagreements, disagreement{utility, variant, fromGenerated, "n/a", "rejected by the generated core's own validator"}) + } + if !handwrittenIsValid(fromGenerated) { + disagreements = append(disagreements, disagreement{utility, variant, fromGenerated, "rejected by its own port's validator", "valid (own validator)"}) + } + } +} + +func compareGenerate(utility string, handwrittenGenerate func() string, generatedGenerate func() string) { + variant := "generate" + fmt.Printf("%s (%s)\n", utility, variant) + handwrittenMs := measure(func() { handwrittenGenerate() }) + fmt.Printf(" handwritten %.1f ms\n", handwrittenMs) + generatedMs := measure(func() { generatedGenerate() }) + fmt.Printf(" generated %.1f ms\n", generatedMs) + rows = append(rows, row{utility, variant, handwrittenMs, generatedMs, iterations}) +} + +func main() { + compareBool("isValidCpf", "full-pipeline", rawCPFs, hcpf.IsValid, core.IsValidCpf) + compareBool("isValidCpf", "normalized", normalizedCPFs, hcpf.IsValid, core.IsValidCpf) + + compareBool("isValidCnpj", "full-pipeline", rawCNPJs, hcnpj.IsValid, func(v string) bool { return core.IsValidCnpj(v, "1") }) + compareBool("isValidCnpj", "normalized", normalizedCNPJs, hcnpj.IsValid, func(v string) bool { return core.IsValidCnpj(v, "1") }) + + compareString("formatCnpj", "full-pipeline", rawCNPJs, hcnpj.Format, func(v string) string { + return core.FormatCnpj(v, core.FormatCnpjOptions{Pad: false, Version: "1", Obfuscate: false}) + }) + + compareCurrency("formatCurrency", "normalized", currencyValues, hcurrency.FormatCurrency, func(v float64) string { + return core.FormatCurrency(toCents(v), true) + }) + + env := benchCapabilities{} // built once, like a real caller would, then reused + // cnpj.Generate(0) defaults to branch=1 (fixed), not a random branch like the generated core -- + // irrelevant here, since only validity is being checked, not equality; see the README. + checkGeneratorAgreement("generateCpf", "generate", hcpf.Generate, hcpf.IsValid, func() string { return core.GenerateCpf(env) }, core.IsValidCpf) + compareGenerate("generateCpf", hcpf.Generate, func() string { return core.GenerateCpf(env) }) + + checkGeneratorAgreement("generateCnpj", "generate", func() string { return hcnpj.Generate(0) }, hcnpj.IsValid, func() string { return core.GenerateCnpj(env) }, func(v string) bool { return core.IsValidCnpj(v, "1") }) + compareGenerate("generateCnpj", func() string { return hcnpj.Generate(0) }, func() string { return core.GenerateCnpj(env) }) + + fmt.Println("\n| utility | variant | handwritten | generated | ratio | budget |") + fmt.Println("| --- | --- | --- | --- | --- | --- |") + for _, r := range rows { + ratio := r.GeneratedMs / r.HandwrittenMs + status := "within" + if ratio > budget { + status = "OVER" + } + fmt.Printf("| `%s` | %s | %.1f ms | %.1f ms | %.2fx | %s %.1fx |\n", r.Utility, r.Variant, r.HandwrittenMs, r.GeneratedMs, ratio, status, budget) + } + + if len(disagreements) > 0 { + fmt.Println("\nDISAGREEMENTS:") + for _, d := range disagreements { + fmt.Printf(" %s (%s) input=%v handwritten=%v generated=%v\n", d.Utility, d.Variant, d.Input, d.Handwritten, d.Generated) + } + } + + skipped := []map[string]string{ + { + "utility": "formatCnpj (obfuscate/pad/version 2)", + "reason": "the Go port's Format has no pad, obfuscate or alphanumeric option, so only the plain full-pipeline case is comparable", + }, + { + "utility": "getHolidays", + "reason": "brazilian-utils/go's date package has no getHolidays: only date.IsHoliday(time.Time, uf), a single-day boolean check, not a function returning a year's list. No comparable counterpart.", + }, + { + "utility": "isBusinessDay", + "reason": "brazilian-utils/go has no isBusinessDay or business-day/weekend concept at all -- only date.IsHoliday, which does not consider weekends. No comparable counterpart.", + }, + } + fmt.Println("\nSKIPPED:") + for _, s := range skipped { + fmt.Printf(" %s: %s\n", s["utility"], s["reason"]) + } + + result := map[string]interface{}{ + "language": "go", + "toolchain": map[string]string{ + "go": runtime.Version(), + }, + "rows": rows, + "disagreements": disagreements, + "skipped": skipped, + } + out, err := json.Marshal(result) + if err != nil { + fmt.Fprintln(os.Stderr, err) + os.Exit(1) + } + fmt.Printf("BENCH_JSON %s\n", out) +} diff --git a/core/bench/go/go.mod b/core/bench/go/go.mod new file mode 100644 index 000000000..ed384782f --- /dev/null +++ b/core/bench/go/go.mod @@ -0,0 +1,14 @@ +module benchgo + +go 1.23.5 + +toolchain go1.24.7 + +require ( + coreout v0.0.0-00010101000000-000000000000 + github.com/brazilian-utils/go v0.0.0-00010101000000-000000000000 +) + +replace coreout => ../../out/go + +replace github.com/brazilian-utils/go => ../../../../brazilian-utils/go diff --git a/core/bench/python.py b/core/bench/python.py new file mode 100644 index 000000000..2335a6cae --- /dev/null +++ b/core/bench/python.py @@ -0,0 +1,370 @@ +#!/usr/bin/env python3 +"""Benchmarks the generated Python core against brazilian-utils' handwritten package. + +`brutils/cpf.py` and `brutils/cnpj.py` are loaded directly by file path rather than imported as +the `brutils` package, because the package's `pyproject.toml` declares `holidays` and `num2words` +that those two modules never actually import -- see core/bench/README.md. + +Fairness note: unlike the JS handwritten package, `brutils.cpf.is_valid` / `brutils.cnpj.is_valid` +do **not** strip mask characters themselves -- they require an already-digit (or, for CNPJ, +already-alphanumeric) string and return False on anything else without doing any real validation +work. Feeding them a masked string like "123.456.789-09" would not exercise their validation logic +at all, so it would not be a real speed comparison. Every row here is therefore "normalized-only": +both sides receive the same pre-sanitized, mask-free string. This means the generated core is +still doing marginally more work than strictly necessary (its regex still walks the string +checking for mask characters that are not there), which is reported plainly rather than hidden -- +see the README. + +formatCnpj is left out for Python entirely: `brutils.cnpj.format_cnpj` calls `is_valid` first and +returns None for a bad checksum, while the generated `formatCnpj` never validates and has no +concept of a bad checksum. That is not a normalization difference to paper over with a shared +input -- it is a different contract (validate-then-format vs. format-unconditionally), so timing +them against each other would mostly measure the checksum computation Python's port does and ours +does not. + +formatCurrency is covered, but needs `num2words` installed (`pip install num2words`): +`brutils/currency.py` imports it at module scope even though `format_currency` never calls it, and +unlike cpf.py/cnpj.py's unused `holidays` import, this one cannot be dodged by loading the file +directly -- the import is inside currency.py itself. getHolidays and isBusinessDay are *not* +covered for Python: brutils has no counterpart to either (only `is_holiday(date, uf)`, a +single-day boolean check with no year-list form and no weekend/business-day concept at all); see +the `skipped` entries below and core/bench/README.md for the full reasoning. + +generateCpf and generateCnpj are covered by the "does every value validate" rule described in +core/bench/README.md, not by equality (both sides draw at random). Both are called the same way +the handwritten port is, with no arguments: `python._support.DEFAULT_CAPABILITIES`, the generated +core's own real (non-fixture) environment, is what a bare `generate_cpf()`/`generate_cnpj()` call +now uses internally, built once at import time rather than per call. See the README for what its +`next_u32` -- `random.getrandbits`, not a CSPRNG -- costs against brutils' own `random` module use. +""" + +import importlib.util +import json +import os +import platform +import sys +import time +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[2] +# The handwritten ports are separate repositories, checked out beside this one by default. +# `BRUTILS_ROOT` overrides that for a checkout that lives somewhere else. +PORTS_ROOT = Path(os.environ.get("BRUTILS_ROOT", REPO_ROOT.parent / "brazilian-utils")) +BRUTILS_ROOT = PORTS_ROOT / "python" / "brutils" + +BUDGET = 1.5 +WARMUP = 20_000 +ITERATIONS = 200_000 + + +def load_module_by_path(name: str, path: Path): + if not path.exists(): + sys.stderr.write( + f"error: missing handwritten Python port at {path}\n" + "Clone it first:\n" + " git clone https://github.com/brazilian-utils/brazilian-utils /home/user/brazilian-utils " + "(or clone the python/ subtree separately -- see core/bench/README.md)\n" + ) + sys.exit(1) + spec = importlib.util.spec_from_file_location(name, path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +handwritten_cpf = load_module_by_path("brutils_cpf", BRUTILS_ROOT / "cpf.py") +handwritten_cnpj = load_module_by_path("brutils_cnpj", BRUTILS_ROOT / "cnpj.py") + +# currency.py imports num2words at module scope even though format_currency itself never calls it +# -- the same "declared but not used by this function" situation cpf.py/cnpj.py would have with +# `holidays`, except this one cannot be dodged by loading the file directly: the import is inside +# currency.py itself, not __init__.py's re-export chain. `pip install num2words` is required for +# this row; see core/bench/README.md. +handwritten_currency = load_module_by_path("brutils_currency", BRUTILS_ROOT / "currency.py") + +# The generated core is a normal package (core/out/python/__init__.py exists), so it is imported +# the ordinary way once core/out is on sys.path. +sys.path.insert(0, str(REPO_ROOT / "core" / "out")) +from python import is_valid_cpf as generated_cpf_module # noqa: E402 +from python import is_valid_cnpj as generated_cnpj_module # noqa: E402 +from python import format_currency as generated_currency_module # noqa: E402 +from python import generate_cpf as generated_generate_cpf_module # noqa: E402 +from python import generate_cnpj as generated_generate_cnpj_module # noqa: E402 + +generated_cpf = generated_cpf_module.is_valid_cpf +generated_cnpj = generated_cnpj_module.is_valid_cnpj +generated_format_currency = generated_currency_module.format_currency +generated_generate_cpf = generated_generate_cpf_module.generate_cpf +generated_generate_cnpj = generated_generate_cnpj_module.generate_cnpj + +# Mask-free, since brutils' own validators require that shape (see module docstring above). +NORMALIZED_CPFS = ["12345678909", "00000000000", "52998224725", "11144477735"] +NORMALIZED_CNPJS = ["12345678000195", "00000000000000", "Q0SLFMBD7VX439"] + +# Raw floats, the shape brutils.currency.format_currency's own callers use. Includes a negative +# value on purpose -- see the README's "formatCurrency" honesty note for what that turns up. +CURRENCY_VALUES = [0, 1234.56, -1234.56, 0.5, 999999.99, 10] + + +def to_cents(value): + """The scaled integer (Decimal<2>) the generated core's formatCurrency requires.""" + return round(value * 100) + + +rows = [] +disagreements = [] +skipped = [ + { + "utility": "formatCnpj", + "reason": ( + "brutils.cnpj.format_cnpj validates the checksum and returns None on a bad one; the " + "generated formatCnpj never validates. Different contracts, not a fair timing comparison." + ), + }, + { + "utility": "getHolidays", + "reason": ( + "brutils has no getHolidays: date_utils.py exposes only is_holiday(date, uf), a " + "single-day boolean check built on the third-party `holidays` package, not a " + "function that returns a year's list. There is no counterpart with the same shape " + "to compare against, so the row is left out rather than comparing two different " + "operations." + ), + }, + { + "utility": "isBusinessDay", + "reason": ( + "brutils has no isBusinessDay or business-day/weekend concept at all -- only " + "is_holiday(date, uf), which does not consider weekends and has no includeOptional " + "equivalent. Not a fair comparison, so it is left out." + ), + } +] + + +def check_agreement(utility, variant, inputs, handwritten, generated): + for value in inputs: + a = handwritten(value) + b = generated(value) + if a != b: + disagreements.append( + {"utility": utility, "variant": variant, "input": value, "handwritten": a, "generated": b} + ) + + +def measure(run): + for _ in range(WARMUP): + run() + started = time.perf_counter() + for _ in range(ITERATIONS): + run() + return (time.perf_counter() - started) * 1000.0 + + +def compare(utility, variant, inputs, handwritten, generated): + check_agreement(utility, variant, inputs, handwritten, generated) + + print(f"{utility} ({variant})") + cursor = 0 + + def run_handwritten(): + nonlocal cursor + handwritten(inputs[cursor % len(inputs)]) + cursor += 1 + + handwritten_ms = measure(run_handwritten) + print(f" handwritten {handwritten_ms:.1f} ms") + + cursor = 0 + + def run_generated(): + nonlocal cursor + generated(inputs[cursor % len(inputs)]) + cursor += 1 + + generated_ms = measure(run_generated) + print(f" generated {generated_ms:.1f} ms") + + rows.append( + {"utility": utility, "variant": variant, "handwrittenMs": handwritten_ms, "generatedMs": generated_ms, "iterations": ITERATIONS} + ) + + +compare( + "isValidCpf", + "normalized", + NORMALIZED_CPFS, + handwritten_cpf.is_valid, + lambda value: generated_cpf(value), +) + +compare( + "isValidCnpj", + "normalized", + NORMALIZED_CNPJS, + handwritten_cnpj.is_valid, + lambda value: generated_cnpj(value, "2"), +) + +# formatCurrency has no full-pipeline shape to compare: the generated core's contract +# (docs/contracts.md) always takes an already-scaled Decimal<2>, never a raw float, so scaling is +# always done outside the core. "normalized" for the same reason isValidCpf/isValidCnpj are above: +# both sides receive the value pre-processed into the shape their own API expects. brutils.currency +# also has no `symbol` option -- it always prefixes "R$" -- so the generated side is called with +# symbol=True to match. +_currency_utility = "formatCurrency" +_currency_variant = "normalized" +for _value in CURRENCY_VALUES: + _a = handwritten_currency.format_currency(_value) + _b = generated_format_currency(to_cents(_value), True) + if _a != _b: + disagreements.append( + {"utility": _currency_utility, "variant": _currency_variant, "input": _value, "handwritten": _a, "generated": _b} + ) + +print(f"{_currency_utility} ({_currency_variant})") +_cursor = 0 + + +def _run_handwritten_currency(): + global _cursor + handwritten_currency.format_currency(CURRENCY_VALUES[_cursor % len(CURRENCY_VALUES)]) + _cursor += 1 + + +_handwritten_currency_ms = measure(_run_handwritten_currency) +print(f" handwritten {_handwritten_currency_ms:.1f} ms") +_cursor = 0 + + +def _run_generated_currency(): + global _cursor + generated_format_currency(to_cents(CURRENCY_VALUES[_cursor % len(CURRENCY_VALUES)]), True) + _cursor += 1 + + +_generated_currency_ms = measure(_run_generated_currency) +print(f" generated {_generated_currency_ms:.1f} ms") +rows.append( + { + "utility": _currency_utility, + "variant": _currency_variant, + "handwrittenMs": _handwritten_currency_ms, + "generatedMs": _generated_currency_ms, + "iterations": ITERATIONS, + } +) + +# generateCpf / generateCnpj draw at random, so there is no fixed value to compare for equality. +# The agreement check instead: every value either side produces must validate under BOTH +# validators -- its own port's and the generated core's -- before either side is timed. +GENERATE_SAMPLES = 500 + + +def check_generator_agreement(utility, variant, handwritten_generate, handwritten_is_valid, generated_generate, generated_is_valid): + for _ in range(GENERATE_SAMPLES): + from_handwritten = handwritten_generate() + if not handwritten_is_valid(from_handwritten): + disagreements.append( + { + "utility": utility, + "variant": variant, + "input": from_handwritten, + "handwritten": "rejected by its own port's validator", + "generated": "n/a", + } + ) + if not generated_is_valid(from_handwritten): + disagreements.append( + { + "utility": utility, + "variant": variant, + "input": from_handwritten, + "handwritten": "valid (own validator)", + "generated": "rejected by the generated core's validator", + } + ) + + from_generated = generated_generate() + if not generated_is_valid(from_generated): + disagreements.append( + { + "utility": utility, + "variant": variant, + "input": from_generated, + "handwritten": "n/a", + "generated": "rejected by the generated core's own validator", + } + ) + if not handwritten_is_valid(from_generated): + disagreements.append( + { + "utility": utility, + "variant": variant, + "input": from_generated, + "handwritten": "rejected by its own port's validator", + "generated": "valid (own validator)", + } + ) + + +def compare_generate(utility, handwritten_generate, generated_generate): + variant = "generate" + print(f"{utility} ({variant})") + handwritten_ms = measure(handwritten_generate) + print(f" handwritten {handwritten_ms:.1f} ms") + generated_ms = measure(generated_generate) + print(f" generated {generated_ms:.1f} ms") + rows.append({"utility": utility, "variant": variant, "handwrittenMs": handwritten_ms, "generatedMs": generated_ms, "iterations": ITERATIONS}) + + +# brutils.cnpj.generate defaults to branch=1 (fixed), not a random branch like the generated core +# -- irrelevant here, since only validity is being checked, not equality; see the README. +check_generator_agreement( + "generateCpf", + "generate", + handwritten_cpf.generate, + handwritten_cpf.is_valid, + lambda: generated_generate_cpf(), + generated_cpf, +) +compare_generate("generateCpf", handwritten_cpf.generate, lambda: generated_generate_cpf()) + +check_generator_agreement( + "generateCnpj", + "generate", + handwritten_cnpj.generate, + handwritten_cnpj.is_valid, + lambda: generated_generate_cnpj(), + lambda value: generated_cnpj(value, "1"), +) +compare_generate("generateCnpj", handwritten_cnpj.generate, lambda: generated_generate_cnpj()) + +print("\n| utility | variant | handwritten | generated | ratio | budget |") +print("| --- | --- | --- | --- | --- | --- |") +for row in rows: + ratio = row["generatedMs"] / row["handwrittenMs"] + ok = ratio <= BUDGET + status = "within" if ok else "OVER" + print( + f"| `{row['utility']}` | {row['variant']} | {row['handwrittenMs']:.1f} ms | {row['generatedMs']:.1f} ms | {ratio:.2f}x | {status} {BUDGET}x |" + ) + +if disagreements: + print("\nDISAGREEMENTS:") + for d in disagreements: + print(f" {d['utility']} ({d['variant']}) input={d['input']!r} handwritten={d['handwritten']!r} generated={d['generated']!r}") + +if skipped: + print("\nSKIPPED:") + for s in skipped: + print(f" {s['utility']}: {s['reason']}") + +result = { + "language": "python", + "toolchain": {"python": platform.python_version()}, + "rows": rows, + "disagreements": disagreements, + "skipped": skipped, +} +print(f"BENCH_JSON {json.dumps(result)}") diff --git a/core/bench/run.mjs b/core/bench/run.mjs new file mode 100644 index 000000000..e9da241cb --- /dev/null +++ b/core/bench/run.mjs @@ -0,0 +1,218 @@ +#!/usr/bin/env node +/** + * Combined cross-language benchmark runner. + * + * Spawns one subprocess per language. Each subprocess runs entirely in its own language runtime + * (node for TypeScript, python3 for Python, `go run` for Go) and times itself internally -- this + * script never times a call across a process boundary, since that would measure the boundary + * instead of the code. Every harness prints its own human-readable progress and table to stdout, + * plus one trailing line starting with `BENCH_JSON ` carrying a machine-readable summary; this + * script parses only that line and folds every language's rows into one table. + * + * Usage: node core/bench/run.mjs + */ + +import { spawnSync } from "node:child_process"; +import { existsSync } from "node:fs"; +import { fileURLToPath } from "node:url"; +import { dirname, join } from "node:path"; + +const BENCH_DIR = dirname(fileURLToPath(import.meta.url)); +const CORE_DIR = join(BENCH_DIR, ".."); + +// The handwritten ports are separate repositories, checked out beside this one by default. +// `BRUTILS_ROOT` overrides that for a checkout that lives somewhere else. +const PORTS_ROOT = process.env["BRUTILS_ROOT"] ?? join(CORE_DIR, "..", "..", "brazilian-utils"); +const PYTHON_BRUTILS = join(PORTS_ROOT, "python", "brutils"); +const GO_BRUTILS = join(PORTS_ROOT, "go"); +const RUST_BRUTILS = join(PORTS_ROOT, "rust"); +const RUST_HARNESS = join(BENCH_DIR, "rust"); + +function section(title) { + console.log(`\n=== ${title} ===\n`); +} + +function runLanguage(name, { precondition, run }) { + section(name); + if (precondition) { + const problem = precondition(); + if (problem) { + console.log(problem); + return { language: name.toLowerCase(), status: "skipped", reason: problem }; + } + } + + const result = run(); + process.stdout.write(result.stdout ?? ""); + if (result.stderr) process.stderr.write(result.stderr); + + if (result.status !== 0 && result.status !== null) { + return { + language: name.toLowerCase(), + status: "error", + reason: `subprocess exited with code ${result.status}`, + }; + } + + const jsonLine = (result.stdout ?? "") + .split("\n") + .reverse() + .find((line) => line.startsWith("BENCH_JSON ")); + if (!jsonLine) { + return { language: name.toLowerCase(), status: "error", reason: "no BENCH_JSON line in output" }; + } + + try { + const parsed = JSON.parse(jsonLine.slice("BENCH_JSON ".length)); + return { status: "ok", ...parsed }; + } catch (error) { + return { language: name.toLowerCase(), status: "error", reason: `could not parse BENCH_JSON: ${error.message}` }; + } +} + +const results = []; + +results.push( + runLanguage("TypeScript", { + run: () => + spawnSync( + "node", + ["--import", "./conformance/sloppy-imports.mjs", "./bench/typescript.ts"], + { cwd: CORE_DIR, encoding: "utf8" }, + ), + }), +); + +results.push( + runLanguage("Python", { + precondition: () => { + if (!existsSync(PYTHON_BRUTILS)) { + return ( + `Missing handwritten Python port at ${PYTHON_BRUTILS}\n` + + "Clone it first:\n" + + " git clone https://github.com/brazilian-utils/python /home/user/brazilian-utils/python" + ); + } + return null; + }, + run: () => spawnSync("python3", [join(BENCH_DIR, "python.py")], { encoding: "utf8" }), + }), +); + +results.push( + runLanguage("Go", { + precondition: () => { + if (!existsSync(GO_BRUTILS)) { + return ( + `Missing handwritten Go port at ${GO_BRUTILS}\n` + + "Clone it first:\n" + + " git clone https://github.com/brazilian-utils/go /home/user/brazilian-utils/go" + ); + } + return null; + }, + run: () => { + const goDir = join(BENCH_DIR, "go"); + // `go.mod` cannot read an environment variable, and its committed `replace` is a path + // relative to itself, which is right for the default layout. Honour `BRUTILS_ROOT` by + // rewriting the directive for the run and restoring it afterwards, so a checkout + // somewhere else does not leave the file modified. + const DEFAULT_REPLACE = "../../../../brazilian-utils/go"; + const custom = process.env["BRUTILS_ROOT"] !== undefined; + const edit = (target) => + spawnSync("go", ["mod", "edit", `-replace=github.com/brazilian-utils/go=${target}`], { cwd: goDir }); + if (custom) edit(GO_BRUTILS); + try { + return spawnSync("go", ["run", "."], { cwd: goDir, encoding: "utf8" }); + } finally { + if (custom) edit(DEFAULT_REPLACE); + } + }, + }), +); + +// Rust drops in later, from another agent's work in core/out/rust/. This runner does not wait for +// it and does not write it -- it just notices when both the harness and the generated code exist, +// and otherwise reports why the row is absent instead of silently omitting Rust from the table. +results.push( + runLanguage("Rust", { + precondition: () => { + if (!existsSync(RUST_BRUTILS)) { + return ( + `Missing handwritten Rust port at ${RUST_BRUTILS}\n` + + "Clone it first:\n" + + " git clone https://github.com/BrazilianUtils/rust /home/user/brazilian-utils/rust" + ); + } + if (!existsSync(join(CORE_DIR, "out", "rust"))) { + return "core/out/rust/ does not exist yet -- the Rust target has not been generated."; + } + if (!existsSync(RUST_HARNESS)) { + return "core/bench/rust/ has not been written yet -- no Rust harness to run."; + } + return null; + }, + run: () => spawnSync("cargo", ["run", "--release"], { cwd: RUST_HARNESS, encoding: "utf8" }), + }), +); + +section("Combined result"); + +const toolchainLines = []; +for (const result of results) { + if (result.status === "ok" && result.toolchain) { + for (const [tool, version] of Object.entries(result.toolchain)) { + toolchainLines.push(`- ${tool}: ${version}`); + } + } +} +console.log("Toolchain versions this run was measured with:"); +console.log(toolchainLines.join("\n") || "(none recorded)"); + +console.log("\n| language | utility | variant | handwritten | generated | ratio (gen/hw) | budget |"); +console.log("| --- | --- | --- | --- | --- | --- | --- |"); + +const BUDGET = 1.5; +let anyDisagreement = false; +let anySkipped = false; + +for (const result of results) { + if (result.status !== "ok") continue; + for (const row of result.rows ?? []) { + const ratio = row.generatedMs / row.handwrittenMs; + const ok = ratio <= BUDGET; + console.log( + `| ${result.language} | \`${row.utility}\` | ${row.variant} | ${row.handwrittenMs.toFixed(1)} ms | ${row.generatedMs.toFixed(1)} ms | ${ratio.toFixed(2)}x | ${ok ? "within" : "OVER"} ${BUDGET}x |`, + ); + } + if ((result.disagreements ?? []).length > 0) anyDisagreement = true; + if ((result.skipped ?? []).length > 0) anySkipped = true; +} + +console.log("\nLanguages not in the table above:"); +for (const result of results) { + if (result.status === "skipped") console.log(`- ${result.language}: ${result.reason}`); + if (result.status === "error") console.log(`- ${result.language}: ERROR -- ${result.reason}`); +} + +if (anyDisagreement) { + console.log("\nDISAGREEMENTS (generated vs. handwritten returned different answers -- see above per language):"); + for (const result of results) { + for (const d of result.disagreements ?? []) { + console.log( + `- ${result.language} ${d.utility} (${d.variant}): input=${JSON.stringify(d.input)} handwritten=${JSON.stringify(d.handwritten)} generated=${JSON.stringify(d.generated)}`, + ); + } + } +} else { + console.log("\nNo disagreements: every benchmarked input produced the same answer on both sides."); +} + +if (anySkipped) { + console.log("\nComparisons left out (contract mismatch -- see core/bench/README.md for the full reasoning):"); + for (const result of results) { + for (const s of result.skipped ?? []) { + console.log(`- ${result.language} ${s.utility}: ${s.reason}`); + } + } +} diff --git a/core/bench/rust/Cargo.lock b/core/bench/rust/Cargo.lock new file mode 100644 index 000000000..a5a9ec18a --- /dev/null +++ b/core/bench/rust/Cargo.lock @@ -0,0 +1,1641 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "aho-corasick" +version = "1.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c982642fa9e8606056828ee9a8505737230110bb1099153c79efe865c59d12ba" +dependencies = [ + "memchr", +] + +[[package]] +name = "android_system_properties" +version = "0.1.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae221649c9976a6f6c56ae1facf410f3ddb33cc661c4b7b61020a912d4237fbc" +dependencies = [ + "libc", +] + +[[package]] +name = "autocfg" +version = "1.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f2032f911046de80f0a198e0901378627c33f59ea0ac00e363d481118bd70a53" + +[[package]] +name = "base64" +version = "0.21.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9d297deb1925b89f2ccc13d7635fa0714f12c87adce1c75356b39ca9b7178567" + +[[package]] +name = "benchrust" +version = "0.1.0" +dependencies = [ + "brazilian_utils", + "coreout", +] + +[[package]] +name = "bitflags" +version = "1.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a" + +[[package]] +name = "bitflags" +version = "2.13.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3ded4057c258ba199e2d26386d3af3780957ecaee6c4ef4041c6b4b8b97c0b06" + +[[package]] +name = "brazilian_utils" +version = "0.1.1" +dependencies = [ + "chrono", + "rand", + "regex", + "reqwest", + "serde", + "serde_json", + "unicode-normalization", +] + +[[package]] +name = "bumpalo" +version = "3.20.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" + +[[package]] +name = "bytes" +version = "1.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" + +[[package]] +name = "cc" +version = "1.4.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "54413ede23c2daf518f35156dfde027feb2374004d63bd497f983c8db9c0e313" +dependencies = [ + "find-msvc-tools", + "shlex", +] + +[[package]] +name = "cfg-if" +version = "1.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600" + +[[package]] +name = "chrono" +version = "0.4.45" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" +dependencies = [ + "iana-time-zone", + "js-sys", + "num-traits", + "wasm-bindgen", + "windows-link", +] + +[[package]] +name = "core-foundation" +version = "0.9.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "91e195e091a93c46f7102ec7818a2aa394e1e1771c3ab4825963fa03e45afb8f" +dependencies = [ + "core-foundation-sys", + "libc", +] + +[[package]] +name = "core-foundation" +version = "0.10.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b2a6cd9ae233e7f62ba4e9353e81a88df7fc8a5987b8d445b4d90c879bd156f6" +dependencies = [ + "core-foundation-sys", + "libc", +] + +[[package]] +name = "core-foundation-sys" +version = "0.8.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b" + +[[package]] +name = "core_detect" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7f8f80099a98041a3d1622845c271458a2d73e688351bf3cb999266764b81d48" + +[[package]] +name = "coreout" +version = "0.1.0" + +[[package]] +name = "displaydoc" +version = "0.2.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c6232dd377dcc64799954cbd3a9bb882e9cdc1308ccd87b1c098f1fb2eaf82a8" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "encoding_rs" +version = "0.8.41" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7b5ef0006ac9ab233c38522f5ae99cae3625151de8f706cacee1cba4b8e2832a" +dependencies = [ + "cfg-if", + "core_detect", + "multiversion", + "multiversion_no_op", + "rustversion", + "scopeguard", + "simdutf8", +] + +[[package]] +name = "equivalent" +version = "1.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f" + +[[package]] +name = "errno" +version = "0.3.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb" +dependencies = [ + "libc", + "windows-sys 0.61.2", +] + +[[package]] +name = "fastrand" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" + +[[package]] +name = "find-msvc-tools" +version = "0.1.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ef25905e51abafe4dcea6c15fec58c57b601cdbd0ee53d22ea1d3016c587d39b" + +[[package]] +name = "fnv" +version = "1.0.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3f9eec918d3f24069decb9af1554cad7c880e2da24a9afd88aca000531ab82c1" + +[[package]] +name = "foreign-types" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f6f339eb8adc052cd2ca78910fda869aefa38d22d5cb648e6485e4d3fc06f3b1" +dependencies = [ + "foreign-types-shared", +] + +[[package]] +name = "foreign-types-shared" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "00b0228411908ca8685dba7fc2cdd70ec9990a6e753e89b6ac91a84c40fbaf4b" + +[[package]] +name = "form_urlencoded" +version = "1.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cb4cb245038516f5f85277875cdaa4f7d2c9a0fa0468de06ed190163b1581fcf" +dependencies = [ + "percent-encoding", +] + +[[package]] +name = "futures-channel" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b1f9e3d69d39e4862ffed03ed071a76f9a13ba1d9109d355b0f0aa6b15e393c4" +dependencies = [ + "futures-core", +] + +[[package]] +name = "futures-core" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" + +[[package]] +name = "futures-io" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "53c0fa8157de1303bfffdaa1cc2a673bfffb60102f76b0ef4441659124373fed" + +[[package]] +name = "futures-sink" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1944426bf7d03f1d14f708785e4b33efd750b36d48a157b836b3efc15ede8e1d" + +[[package]] +name = "futures-task" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd" + +[[package]] +name = "futures-util" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc" +dependencies = [ + "futures-core", + "futures-io", + "futures-task", + "memchr", + "pin-project-lite", + "slab", +] + +[[package]] +name = "getrandom" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ff2abc00be7fca6ebc474524697ae276ad847ad0a6b3faa4bcb027e9a4614ad0" +dependencies = [ + "cfg-if", + "libc", + "wasi", +] + +[[package]] +name = "getrandom" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" +dependencies = [ + "cfg-if", + "libc", + "r-efi", +] + +[[package]] +name = "h2" +version = "0.3.27" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0beca50380b1fc32983fc1cb4587bfa4bb9e78fc259aad4a0032d2080309222d" +dependencies = [ + "bytes", + "fnv", + "futures-core", + "futures-sink", + "futures-util", + "http", + "indexmap", + "slab", + "tokio", + "tokio-util", + "tracing", +] + +[[package]] +name = "hashbrown" +version = "0.17.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" + +[[package]] +name = "http" +version = "0.2.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "601cbb57e577e2f5ef5be8e7b83f0f63994f25aa94d673e54a92d5c516d101f1" +dependencies = [ + "bytes", + "fnv", + "itoa", +] + +[[package]] +name = "http-body" +version = "0.4.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7ceab25649e9960c0311ea418d17bee82c0dcec1bd053b5f9a66e265a693bed2" +dependencies = [ + "bytes", + "http", + "pin-project-lite", +] + +[[package]] +name = "httparse" +version = "1.10.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6dbf3de79e51f3d586ab4cb9d5c3e2c14aa28ed23d180cf89b4df0454a69cc87" + +[[package]] +name = "httpdate" +version = "1.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "df3b46402a9d5adb4c86a0cf463f42e19994e3ee891101b1841f30a545cb49a9" + +[[package]] +name = "hyper" +version = "0.14.32" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "41dfc780fdec9373c01bae43289ea34c972e40ee3c9f6b3c8801a35f35586ce7" +dependencies = [ + "bytes", + "futures-channel", + "futures-core", + "futures-util", + "h2", + "http", + "http-body", + "httparse", + "httpdate", + "itoa", + "pin-project-lite", + "socket2 0.5.10", + "tokio", + "tower-service", + "tracing", + "want", +] + +[[package]] +name = "hyper-tls" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d6183ddfa99b85da61a140bea0efc93fdf56ceaa041b37d553518030827f9905" +dependencies = [ + "bytes", + "hyper", + "native-tls", + "tokio", + "tokio-native-tls", +] + +[[package]] +name = "iana-time-zone" +version = "0.1.65" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e31bc9ad994ba00e440a8aa5c9ef0ec67d5cb5e5cb0cc7f8b744a35b389cc470" +dependencies = [ + "android_system_properties", + "core-foundation-sys", + "iana-time-zone-haiku", + "js-sys", + "log", + "wasm-bindgen", + "windows-core", +] + +[[package]] +name = "iana-time-zone-haiku" +version = "0.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f31827a206f56af32e590ba56d5d2d085f558508192593743f16b2306495269f" +dependencies = [ + "cc", +] + +[[package]] +name = "icu_collections" +version = "2.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fa68d21081c4a05d5a901a1c62add574c77048b6a1c67be3b50ce0b60d4ca513" +dependencies = [ + "displaydoc", + "potential_utf", + "utf8_iter", + "yoke", + "zerofrom", + "zerovec", +] + +[[package]] +name = "icu_locale_core" +version = "2.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d56e28588da92eee5c3201a6eff33fabdd49b62269c8938d4ff050ce4d900deb" +dependencies = [ + "displaydoc", + "litemap", + "tinystr", + "writeable", + "zerovec", +] + +[[package]] +name = "icu_normalizer" +version = "2.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "12f9cf5f235641ed274641dd81c3f28d870e276763d0797aeeab72317b1c646f" +dependencies = [ + "icu_collections", + "icu_normalizer_data", + "icu_properties", + "icu_provider", + "smallvec", + "zerovec", +] + +[[package]] +name = "icu_normalizer_data" +version = "2.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1563da1ed3e0b3bf3d74c9b85917ac9c56464d2f57242270c09c9e752f8021a0" + +[[package]] +name = "icu_properties" +version = "2.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7e7ca276ad3145661a65914e6daf131ca5120cd3dcee8f8f3214b8875184a148" +dependencies = [ + "displaydoc", + "icu_collections", + "icu_locale_core", + "icu_properties_data", + "icu_provider", + "zerotrie", + "zerovec", +] + +[[package]] +name = "icu_properties_data" +version = "2.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e590f038c1464a96894fd6d10127e90a8be4509f56ff7ecef851b15cee0b7caa" + +[[package]] +name = "icu_provider" +version = "2.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d27bbb9d3abbefac45d55f647c9de1d44aafcd1186eb91879afef17c396c3e73" +dependencies = [ + "displaydoc", + "icu_locale_core", + "writeable", + "yoke", + "zerofrom", + "zerotrie", + "zerovec", +] + +[[package]] +name = "idna" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3b0875f23caa03898994f6ddc501886a45c7d3d62d04d2d90788d47be1b1e4de" +dependencies = [ + "idna_adapter", + "smallvec", + "utf8_iter", +] + +[[package]] +name = "idna_adapter" +version = "1.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cb68373c0d6620ef8105e855e7745e18b0d00d3bdb07fb532e434244cdb9a714" +dependencies = [ + "icu_normalizer", + "icu_properties", +] + +[[package]] +name = "indexmap" +version = "2.14.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cc4e190f5d26ca7051642629da2c52fc03bde85a03197c99408dcd291734c855" +dependencies = [ + "equivalent", + "hashbrown", +] + +[[package]] +name = "ipnet" +version = "2.12.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "791930b43c0d5973160d90a8f3894509f2b273430f5c5c73b668636d0287c5c0" + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + +[[package]] +name = "js-sys" +version = "0.3.105" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ce57d20d1ea864ce2ac172ab472d409214f4fd359f0b2a2775abdf522e2af99e" +dependencies = [ + "cfg-if", + "futures-util", + "wasm-bindgen", +] + +[[package]] +name = "libc" +version = "0.2.189" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" + +[[package]] +name = "linux-raw-sys" +version = "0.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" + +[[package]] +name = "litemap" +version = "0.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "47d9d19d1d6efa0109d2f65ff4c85cddd50bd572e5a00127ab10987290bcefae" + +[[package]] +name = "log" +version = "0.4.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" + +[[package]] +name = "memchr" +version = "2.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" + +[[package]] +name = "mime" +version = "0.3.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6877bb514081ee2a7ff5ef9de3281f14a4dd4bceac4c09388074a6b5df8a139a" + +[[package]] +name = "mio" +version = "1.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4b18443e9c262bfe8fa82f51666e2642c53393f7e5c27b3e1aeab922cff5b9d8" +dependencies = [ + "libc", + "wasi", + "windows-sys 0.61.2", +] + +[[package]] +name = "multiversion" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b4ca4bea16ffc3f443cf7d866912118196bfef4c6a1556ca00f9f9b00bb43f7c" +dependencies = [ + "multiversion-macros", +] + +[[package]] +name = "multiversion-macros" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0d416831a7317ef4b08bee00b69cbbb9c8763da7959a7026244d6266869f9c83" +dependencies = [ + "proc-macro2", + "quote", + "rustversion", + "syn 3.0.6", +] + +[[package]] +name = "multiversion_no_op" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "743fb55ba31b18fb1ecef6bdc9aa2743314978ac084044301a7eee33fb99a20d" + +[[package]] +name = "native-tls" +version = "0.2.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "465500e14ea162429d264d44189adc38b199b62b1c21eea9f69e4b73cb03bbf2" +dependencies = [ + "libc", + "log", + "openssl", + "openssl-probe", + "openssl-sys", + "schannel", + "security-framework", + "security-framework-sys", + "tempfile", +] + +[[package]] +name = "num-traits" +version = "0.2.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "071dfc062690e90b734c0b2273ce72ad0ffa95f0c74596bc250dcfd960262841" +dependencies = [ + "autocfg", +] + +[[package]] +name = "once_cell" +version = "1.21.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" + +[[package]] +name = "openssl" +version = "0.10.81" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "77823a27f0babb03091cb9ed9ef80af3b39dbc82f97e8fa530374b7dafd87a45" +dependencies = [ + "bitflags 2.13.2", + "cfg-if", + "foreign-types", + "libc", + "openssl-macros", + "openssl-sys", +] + +[[package]] +name = "openssl-macros" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a948666b637a0f465e8564c73e89d4dde00d72d4d473cc972f390fc3dcee7d9c" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "openssl-probe" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7c87def4c32ab89d880effc9e097653c8da5d6ef28e6b539d313baaacfbafcbe" + +[[package]] +name = "openssl-sys" +version = "0.9.117" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b47e7e6bb2c38cd930d25a23b40fa52e068c10e85f3e03a7f5ba5aaca5713695" +dependencies = [ + "cc", + "libc", + "pkg-config", + "vcpkg", +] + +[[package]] +name = "percent-encoding" +version = "2.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" + +[[package]] +name = "pin-project-lite" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" + +[[package]] +name = "pkg-config" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548" + +[[package]] +name = "potential_utf" +version = "0.1.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d83eb9bc6d8e5cf568e7a1101d60ee05e81ed50ea106026f3d18deeb046d7661" +dependencies = [ + "zerovec", +] + +[[package]] +name = "ppv-lite86" +version = "0.2.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85eae3c4ed2f50dcfe72643da4befc30deadb458a9b590d720cde2f2b1e97da9" +dependencies = [ + "zerocopy", +] + +[[package]] +name = "proc-macro2" +version = "1.0.107" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "quote" +version = "1.0.47" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" +dependencies = [ + "proc-macro2", +] + +[[package]] +name = "r-efi" +version = "6.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" + +[[package]] +name = "rand" +version = "0.8.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e058c7de0b26af77780c769414d6257830bb240f3c38477dbc2c16e5f54d6d4c" +dependencies = [ + "libc", + "rand_chacha", + "rand_core", +] + +[[package]] +name = "rand_chacha" +version = "0.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e6c10a63a0fa32252be49d21e7709d4d4baf8d231c2dbce1eaa8141b9b127d88" +dependencies = [ + "ppv-lite86", + "rand_core", +] + +[[package]] +name = "rand_core" +version = "0.6.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ec0be4795e2f6a28069bec0b5ff3e2ac9bafc99e6a9a7dc3547996c5c816922c" +dependencies = [ + "getrandom 0.2.17", +] + +[[package]] +name = "regex" +version = "1.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f020237b6c8eed93db2e2cb53c00c60a8e1bc73da7d073199a1180401450218d" +dependencies = [ + "aho-corasick", + "memchr", + "regex-automata", + "regex-syntax", +] + +[[package]] +name = "regex-automata" +version = "0.4.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ad8553b9b26413251cbf30e620595c7a41b3887f03da04579c0e6b0d6a06b4b2" +dependencies = [ + "aho-corasick", + "memchr", + "regex-syntax", +] + +[[package]] +name = "regex-syntax" +version = "0.8.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4" + +[[package]] +name = "reqwest" +version = "0.11.27" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "dd67538700a17451e7cba03ac727fb961abb7607553461627b97de0b89cf4a62" +dependencies = [ + "base64", + "bytes", + "encoding_rs", + "futures-core", + "futures-util", + "h2", + "http", + "http-body", + "hyper", + "hyper-tls", + "ipnet", + "js-sys", + "log", + "mime", + "native-tls", + "once_cell", + "percent-encoding", + "pin-project-lite", + "rustls-pemfile", + "serde", + "serde_json", + "serde_urlencoded", + "sync_wrapper", + "system-configuration", + "tokio", + "tokio-native-tls", + "tower-service", + "url", + "wasm-bindgen", + "wasm-bindgen-futures", + "web-sys", + "winreg", +] + +[[package]] +name = "rustix" +version = "1.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "891efababe418670775f199f0d233d84843c227a0949a883ce15b37c78d6629d" +dependencies = [ + "bitflags 2.13.2", + "errno", + "libc", + "linux-raw-sys", + "windows-sys 0.61.2", +] + +[[package]] +name = "rustls-pemfile" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1c74cae0a4cf6ccbbf5f359f08efdf8ee7e1dc532573bf0db71968cb56b1448c" +dependencies = [ + "base64", +] + +[[package]] +name = "rustversion" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f" + +[[package]] +name = "ryu" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f" + +[[package]] +name = "schannel" +version = "0.1.29" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "91c1b7e4904c873ef0710c1f407dde2e6287de2bebc1bbbf7d430bb7cbffd939" +dependencies = [ + "windows-sys 0.61.2", +] + +[[package]] +name = "scopeguard" +version = "1.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49" + +[[package]] +name = "security-framework" +version = "3.7.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b7f4bc775c73d9a02cde8bf7b2ec4c9d12743edf609006c7facc23998404cd1d" +dependencies = [ + "bitflags 2.13.2", + "core-foundation 0.10.1", + "core-foundation-sys", + "libc", + "security-framework-sys", +] + +[[package]] +name = "security-framework-sys" +version = "2.17.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ce2691df843ecc5d231c0b14ece2acc3efb62c0a398c7e1d875f3983ce020e3" +dependencies = [ + "core-foundation-sys", + "libc", +] + +[[package]] +name = "serde" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" +dependencies = [ + "serde_core", + "serde_derive", +] + +[[package]] +name = "serde_core" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" +dependencies = [ + "serde_derive", +] + +[[package]] +name = "serde_derive" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + +[[package]] +name = "serde_urlencoded" +version = "0.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d3491c14715ca2294c4d6a88f15e84739788c1d030eed8c110436aafdaa2f3fd" +dependencies = [ + "form_urlencoded", + "itoa", + "ryu", + "serde", +] + +[[package]] +name = "shlex" +version = "2.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" + +[[package]] +name = "simdutf8" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e" + +[[package]] +name = "slab" +version = "0.4.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" + +[[package]] +name = "smallvec" +version = "1.16.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ba467056f1b547ed52077911161fc86985becbc60e8e1857c8a144dab0def891" + +[[package]] +name = "socket2" +version = "0.5.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e22376abed350d73dd1cd119b57ffccad95b4e585a7cda43e286245ce23c0678" +dependencies = [ + "libc", + "windows-sys 0.52.0", +] + +[[package]] +name = "socket2" +version = "0.6.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c3d1e2c7f27f8d4cb10542a02c49005dbd6e93095799d6f3be745fae9f8fedd4" +dependencies = [ + "libc", + "windows-sys 0.61.2", +] + +[[package]] +name = "stable_deref_trait" +version = "1.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ce2be8dc25455e1f91df71bfa12ad37d7af1092ae736f3a6cd0e37bc7810596" + +[[package]] +name = "syn" +version = "2.0.119" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "syn" +version = "3.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "sync_wrapper" +version = "0.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2047c6ded9c721764247e62cd3b03c09ffc529b2ba5b10ec482ae507a4a70160" + +[[package]] +name = "synstructure" +version = "0.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "901704edd0dfe137f1987838ee4f259e4e063c31371bdb423f7ae38ec6f77f02" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "system-configuration" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ba3a3adc5c275d719af8cb4272ea1c4a6d668a777f37e115f6d11ddbc1c8e0e7" +dependencies = [ + "bitflags 1.3.2", + "core-foundation 0.9.4", + "system-configuration-sys", +] + +[[package]] +name = "system-configuration-sys" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a75fb188eb626b924683e3b95e3a48e63551fcfb51949de2f06a9d91dbee93c9" +dependencies = [ + "core-foundation-sys", + "libc", +] + +[[package]] +name = "tempfile" +version = "3.27.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" +dependencies = [ + "fastrand", + "getrandom 0.4.3", + "once_cell", + "rustix", + "windows-sys 0.61.2", +] + +[[package]] +name = "tinystr" +version = "0.8.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b1e27c91459209c2986af3dcf603a5a74a4368754ce37414f59acc971167f643" +dependencies = [ + "displaydoc", + "zerovec", +] + +[[package]] +name = "tinyvec" +version = "1.13.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fd3ca314f692efd6c868f8408f53fe444634a845f96c028b97d35f6a1f79f0ee" + +[[package]] +name = "tokio" +version = "1.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" +dependencies = [ + "bytes", + "libc", + "mio", + "pin-project-lite", + "socket2 0.6.5", + "windows-sys 0.61.2", +] + +[[package]] +name = "tokio-native-tls" +version = "0.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bbae76ab933c85776efabc971569dd6119c580d8f5d448769dec1764bf796ef2" +dependencies = [ + "native-tls", + "tokio", +] + +[[package]] +name = "tokio-util" +version = "0.7.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "494815d09bf52b5548659851081238f0ca39ff638363907596da739561c62c52" +dependencies = [ + "bytes", + "futures-core", + "futures-sink", + "libc", + "pin-project-lite", + "tokio", +] + +[[package]] +name = "tower-service" +version = "0.3.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8df9b6e13f2d32c91b9bd719c00d1958837bc7dec474d94952798cc8e69eeec3" + +[[package]] +name = "tracing" +version = "0.1.44" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "63e71662fa4b2a2c3a26f570f037eb95bb1f85397f3cd8076caed2f026a6d100" +dependencies = [ + "pin-project-lite", + "tracing-core", +] + +[[package]] +name = "tracing-core" +version = "0.1.36" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "db97caf9d906fbde555dd62fa95ddba9eecfd14cb388e4f491a66d74cd5fb79a" +dependencies = [ + "once_cell", +] + +[[package]] +name = "try-lock" +version = "0.2.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" + +[[package]] +name = "unicode-ident" +version = "1.0.26" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954" + +[[package]] +name = "unicode-normalization" +version = "0.1.25" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5fd4f6878c9cb28d874b009da9e8d183b5abc80117c40bbd187a1fde336be6e8" +dependencies = [ + "tinyvec", +] + +[[package]] +name = "url" +version = "2.5.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ff67a8a4397373c3ef660812acab3268222035010ab8680ec4215f38ba3d0eed" +dependencies = [ + "form_urlencoded", + "idna", + "percent-encoding", + "serde", +] + +[[package]] +name = "utf8_iter" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be" + +[[package]] +name = "vcpkg" +version = "0.2.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426" + +[[package]] +name = "want" +version = "0.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bfa7760aed19e106de2c7c0b581b509f2f25d3dacaf737cb82ac61bc6d760b0e" +dependencies = [ + "try-lock", +] + +[[package]] +name = "wasi" +version = "0.11.1+wasi-snapshot-preview1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b" + +[[package]] +name = "wasm-bindgen" +version = "0.2.128" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "aecb87a33d3b0c5e3b7aa46336eaf486cffafbd281b195e4c8b80d50df2351bf" +dependencies = [ + "cfg-if", + "once_cell", + "rustversion", + "wasm-bindgen-macro", + "wasm-bindgen-shared", +] + +[[package]] +name = "wasm-bindgen-futures" +version = "0.4.78" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ef4c5d3d2cdf5c54f4231181768f5510842e350db025faf1f7163b1030ed928" +dependencies = [ + "js-sys", + "wasm-bindgen", +] + +[[package]] +name = "wasm-bindgen-macro" +version = "0.2.128" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a690d511e3c1a8b3a55e33511e3c2c00c78415cd23650f32b808627f5696b9ed" +dependencies = [ + "quote", + "wasm-bindgen-macro-support", +] + +[[package]] +name = "wasm-bindgen-macro-support" +version = "0.2.128" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "411e4887f0071ef2d2164a9d5fdf2d20efbef78fccd3a78b0c10a1dc5295e48a" +dependencies = [ + "bumpalo", + "proc-macro2", + "quote", + "syn 3.0.6", + "wasm-bindgen-shared", +] + +[[package]] +name = "wasm-bindgen-shared" +version = "0.2.128" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "81941cd78d0c92026c33e5e01312845a4cb1e9af3407f9134b100dd03144103e" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "web-sys" +version = "0.3.105" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9fbddc4a036f00ec4f18c83445bd3115cb306a91da554919a099d9222fe4a7f8" +dependencies = [ + "js-sys", + "wasm-bindgen", +] + +[[package]] +name = "windows-core" +version = "0.62.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b8e83a14d34d0623b51dce9581199302a221863196a1dde71a7663a4c2be9deb" +dependencies = [ + "windows-implement", + "windows-interface", + "windows-link", + "windows-result", + "windows-strings", +] + +[[package]] +name = "windows-implement" +version = "0.60.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "windows-interface" +version = "0.59.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "windows-link" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" + +[[package]] +name = "windows-result" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7781fa89eaf60850ac3d2da7af8e5242a5ea78d1a11c49bf2910bb5a73853eb5" +dependencies = [ + "windows-link", +] + +[[package]] +name = "windows-strings" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7837d08f69c77cf6b07689544538e017c1bfcf57e34b4c0ff58e6c2cd3b37091" +dependencies = [ + "windows-link", +] + +[[package]] +name = "windows-sys" +version = "0.48.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "677d2418bec65e3338edb076e806bc1ec15693c5d0104683f2efe857f61056a9" +dependencies = [ + "windows-targets 0.48.5", +] + +[[package]] +name = "windows-sys" +version = "0.52.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d" +dependencies = [ + "windows-targets 0.52.6", +] + +[[package]] +name = "windows-sys" +version = "0.61.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" +dependencies = [ + "windows-link", +] + +[[package]] +name = "windows-targets" +version = "0.48.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9a2fa6e2155d7247be68c096456083145c183cbbbc2764150dda45a87197940c" +dependencies = [ + "windows_aarch64_gnullvm 0.48.5", + "windows_aarch64_msvc 0.48.5", + "windows_i686_gnu 0.48.5", + "windows_i686_msvc 0.48.5", + "windows_x86_64_gnu 0.48.5", + "windows_x86_64_gnullvm 0.48.5", + "windows_x86_64_msvc 0.48.5", +] + +[[package]] +name = "windows-targets" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9b724f72796e036ab90c1021d4780d4d3d648aca59e491e6b98e725b84e99973" +dependencies = [ + "windows_aarch64_gnullvm 0.52.6", + "windows_aarch64_msvc 0.52.6", + "windows_i686_gnu 0.52.6", + "windows_i686_gnullvm", + "windows_i686_msvc 0.52.6", + "windows_x86_64_gnu 0.52.6", + "windows_x86_64_gnullvm 0.52.6", + "windows_x86_64_msvc 0.52.6", +] + +[[package]] +name = "windows_aarch64_gnullvm" +version = "0.48.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2b38e32f0abccf9987a4e3079dfb67dcd799fb61361e53e2882c3cbaf0d905d8" + +[[package]] +name = "windows_aarch64_gnullvm" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32a4622180e7a0ec044bb555404c800bc9fd9ec262ec147edd5989ccd0c02cd3" + +[[package]] +name = "windows_aarch64_msvc" +version = "0.48.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "dc35310971f3b2dbbf3f0690a219f40e2d9afcf64f9ab7cc1be722937c26b4bc" + +[[package]] +name = "windows_aarch64_msvc" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "09ec2a7bb152e2252b53fa7803150007879548bc709c039df7627cabbd05d469" + +[[package]] +name = "windows_i686_gnu" +version = "0.48.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a75915e7def60c94dcef72200b9a8e58e5091744960da64ec734a6c6e9b3743e" + +[[package]] +name = "windows_i686_gnu" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8e9b5ad5ab802e97eb8e295ac6720e509ee4c243f69d781394014ebfe8bbfa0b" + +[[package]] +name = "windows_i686_gnullvm" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0eee52d38c090b3caa76c563b86c3a4bd71ef1a819287c19d586d7334ae8ed66" + +[[package]] +name = "windows_i686_msvc" +version = "0.48.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f55c233f70c4b27f66c523580f78f1004e8b5a8b659e05a4eb49d4166cca406" + +[[package]] +name = "windows_i686_msvc" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "240948bc05c5e7c6dabba28bf89d89ffce3e303022809e73deaefe4f6ec56c66" + +[[package]] +name = "windows_x86_64_gnu" +version = "0.48.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "53d40abd2583d23e4718fddf1ebec84dbff8381c07cae67ff7768bbf19c6718e" + +[[package]] +name = "windows_x86_64_gnu" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "147a5c80aabfbf0c7d901cb5895d1de30ef2907eb21fbbab29ca94c5b08b1a78" + +[[package]] +name = "windows_x86_64_gnullvm" +version = "0.48.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b7b52767868a23d5bab768e390dc5f5c55825b6d30b86c844ff2dc7414044cc" + +[[package]] +name = "windows_x86_64_gnullvm" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "24d5b23dc417412679681396f2b49f3de8c1473deb516bd34410872eff51ed0d" + +[[package]] +name = "windows_x86_64_msvc" +version = "0.48.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed94fce61571a4006852b7389a063ab983c02eb1bb37b47f8272ce92d06d9538" + +[[package]] +name = "windows_x86_64_msvc" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec" + +[[package]] +name = "winreg" +version = "0.50.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "524e57b2c537c0f9b1e69f1965311ec12182b4122e45035b1508cd24d2adadb1" +dependencies = [ + "cfg-if", + "windows-sys 0.48.0", +] + +[[package]] +name = "writeable" +version = "0.6.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3ad82d2a33cdc9674dc7465672f271e096168fcdbe0f799d9e6db8c5892679dc" + +[[package]] +name = "yoke" +version = "0.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "709fe23a0424b6a435d82152b1bd3fdfb0833487d5fa90d05d42762a9891fef5" +dependencies = [ + "stable_deref_trait", + "yoke-derive", + "zerofrom", +] + +[[package]] +name = "yoke-derive" +version = "0.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "33811428bee40dbceb6d545e95754741d17a6aef9a4849f0fd62e2ba4f412a78" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", + "synstructure", +] + +[[package]] +name = "zerocopy" +version = "0.8.57" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d35102a9f36d089ccae9e4c6802bc118be4487b80aaffc0ab4e0cf5ce92d2873" +dependencies = [ + "zerocopy-derive", +] + +[[package]] +name = "zerocopy-derive" +version = "0.8.57" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "146c01f5ab44258da43cf276c74a2763db2ff3969c9c652c3f2de07041d0b2bc" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "zerofrom" +version = "0.1.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ec05a11813ea801ff6d75110ad09cd0824ddba17dfe17128ea0d5f68e6c5272" +dependencies = [ + "zerofrom-derive", +] + +[[package]] +name = "zerofrom-derive" +version = "0.1.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f75b4683f6c7f45248d4d64056a24298c6281e0993356d7d1b4a1a962ef10d4a" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", + "synstructure", +] + +[[package]] +name = "zerotrie" +version = "0.2.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4ea269c3bd32f0a32c321907a2ae912ba6f4649bb0fc764a15627e99a7095a3f" +dependencies = [ + "displaydoc", + "yoke", + "zerofrom", +] + +[[package]] +name = "zerovec" +version = "0.11.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bb0464e17806c1d976d5cba29399c7f08e516e279e2ba493f63123b5fca67dd8" +dependencies = [ + "yoke", + "zerofrom", + "zerovec-derive", +] + +[[package]] +name = "zerovec-derive" +version = "0.11.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "34df6fc39dbd26ddc9c10e6a2984476e13acce22e64e4487636ef494369225da" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/core/bench/rust/Cargo.toml b/core/bench/rust/Cargo.toml new file mode 100644 index 000000000..ff79245c3 --- /dev/null +++ b/core/bench/rust/Cargo.toml @@ -0,0 +1,19 @@ +# Benchmarks the generated Rust core against brazilian-utils/rust, the handwritten crate it would +# replace. Both are path dependencies: the generated one is produced by `npm run build` in `core/`, +# and the handwritten one is a separate repository checked out beside this one (see +# core/bench/README.md). Run with `cargo run --release` from this directory. +[package] +name = "benchrust" +version = "0.1.0" +edition = "2021" + +[dependencies] +coreout = { path = "../../out/rust" } +brazilian_utils = { path = "../../../../brazilian-utils/rust" } + +[profile.release] +# The handwritten and generated sides must be optimized identically for the comparison to mean +# anything; these are cargo's release defaults, stated rather than assumed. +opt-level = 3 +lto = false +codegen-units = 16 diff --git a/core/bench/rust/build.rs b/core/bench/rust/build.rs new file mode 100644 index 000000000..6a4d9d959 --- /dev/null +++ b/core/bench/rust/build.rs @@ -0,0 +1,16 @@ +//! Records the compiler the numbers were measured with, so the reported toolchain is the real one +//! rather than whatever was written down by hand. + +use std::process::Command; + +fn main() { + let version = Command::new("rustc") + .arg("--version") + .output() + .ok() + .and_then(|out| String::from_utf8(out.stdout).ok()) + .map(|text| text.trim().to_owned()) + .unwrap_or_else(|| "unknown".to_owned()); + println!("cargo:rustc-env=BENCH_RUSTC_VERSION={version}"); + println!("cargo:rerun-if-changed=build.rs"); +} diff --git a/core/bench/rust/src/main.rs b/core/bench/rust/src/main.rs new file mode 100644 index 000000000..bed6ff549 --- /dev/null +++ b/core/bench/rust/src/main.rs @@ -0,0 +1,399 @@ +//! Generated Rust core against the handwritten `brazilian_utils` crate, on the same inputs. +//! +//! `brazilian_utils::cpf::is_valid` and `cnpj::is_valid` reject anything that is not already +//! digits-only before doing any work, which is the same contract the generated core takes, so +//! every row here is the `normalized` variant -- the same shape the Python rows have, and unlike +//! Go, whose port normalizes internally and therefore needs a full-pipeline row as well. +//! +//! `cnpj::format_cnpj` is left out for the reason Python's is: it validates the checksum and +//! answers `None` for a bad one, while the generated `format_cnpj` never validates, so timing +//! them against each other would mostly measure a checksum the generated side does not compute. +//! +//! `formatCurrency` has no full-pipeline shape to compare at all: the generated core's contract +//! (core/docs/contracts.md) always takes an already-scaled `Decimal<2>` `i64`, never a raw `f64` +//! -- scaling is DX work, done once outside the core -- so this row is `normalized` for the same +//! reason isValidCpf/isValidCnpj are: pre-processed input on both sides. +//! +//! `getHolidays` and `isBusinessDay` are not covered here: `brazilian_utils::date_utils` exposes +//! only `is_holiday(NaiveDate, Option<&str>) -> Option`, a single-day boolean check, not a +//! function returning a year's list, and it has no weekend/business-day concept at all. No fair +//! counterpart, so both rows are left out; see `core/bench/README.md`. +//! +//! `generateCpf`/`generateCnpj` draw at random, so there is nothing to compare for equality; see +//! the README for the "does every value validate" rule used instead. `BenchCapabilities::next_u32` +//! below is a small SplitMix64, a fast, real (not fixture-scripted), non-cryptographic generator -- +//! the same kind of generator `rand::thread_rng()` is, which `brazilian_utils::cpf::generate` and +//! `cnpj::generate` already use, so the RNG choice itself is not what any gap here would measure. + +use std::sync::atomic::{AtomicU64, Ordering}; +use std::time::Instant; + +/// The "real" (non-fixture) environment `coreout::generate_cpf`/`generate_cnpj` need. Only +/// `next_u32` is ever called by them; the other three methods are unreachable stubs, since the +/// `Capabilities` trait requires all four. +struct BenchCapabilities { + state: AtomicU64, +} + +impl BenchCapabilities { + fn new() -> Self { + let seed = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_nanos() as u64) + .unwrap_or(0x9E3779B97F4A7C15); + BenchCapabilities { state: AtomicU64::new(seed | 1) } + } +} + +impl coreout::support::Capabilities for BenchCapabilities { + fn request(&self, _request: coreout::support::HttpRequest) -> Option { + panic!("not used by generate_cpf/generate_cnpj") + } + fn now(&self) -> i64 { + panic!("not used by generate_cpf/generate_cnpj") + } + fn sleep(&self, _millis: i64) { + panic!("not used by generate_cpf/generate_cnpj") + } + fn next_u32(&self) -> i64 { + // SplitMix64, run once per call. + let mut z = self + .state + .fetch_add(0x9E3779B97F4A7C15, Ordering::Relaxed) + .wrapping_add(0x9E3779B97F4A7C15); + z = (z ^ (z >> 30)).wrapping_mul(0xBF58476D1CE4E5B9); + z = (z ^ (z >> 27)).wrapping_mul(0x94D049BB133111EB); + z ^= z >> 31; + (z & 0xFFFF_FFFF) as i64 + } +} + +const WARMUP: usize = 20_000; +const ITERATIONS: usize = 200_000; + +/// Already digits-only: what both sides' validators are documented to take. +const CPFS: [&str; 5] = [ + "12345678909", + "52998224725", + "00000000000", + "11111111111", + "12345678900", +]; + +const CNPJS: [&str; 4] = [ + "12345678000195", + "11222333000181", + "00000000000000", + "12345678000100", +]; + +/// Raw doubles, the shape `brazilian_utils::currency::format_currency`'s own callers use. +/// Includes a negative value on purpose -- see the README's "formatCurrency" honesty note. +const CURRENCY_VALUES: [f64; 6] = [0.0, 1234.56, -1234.56, 0.5, 999999.99, 10.0]; + +/// The scaled integer (`Decimal<2>`) the generated core's `format_currency` requires. +fn to_cents(value: f64) -> i64 { + (value * 100.0).round() as i64 +} + +const GENERATE_SAMPLES: usize = 500; + +struct Row { + utility: &'static str, + variant: &'static str, + handwritten_ms: f64, + generated_ms: f64, +} + +struct Disagreement { + utility: &'static str, + variant: &'static str, + input: String, + handwritten: String, + generated: String, +} + +fn measure(mut run: impl FnMut(usize)) -> f64 { + for index in 0..WARMUP { + run(index); + } + let started = Instant::now(); + for index in 0..ITERATIONS { + run(index); + } + started.elapsed().as_secs_f64() * 1000.0 +} + +/// Times both sides of one comparison, but only after they agree on every input: a ratio between +/// two functions that answer differently is not a measurement of anything. +fn compare( + rows: &mut Vec, + disagreements: &mut Vec, + utility: &'static str, + variant: &'static str, + inputs: &[&str], + handwritten: impl Fn(&str) -> bool, + generated: impl Fn(&str) -> bool, +) { + for input in inputs { + let left = handwritten(input); + let right = generated(input); + if left != right { + disagreements.push(Disagreement { + utility, + variant, + input: (*input).to_owned(), + handwritten: left.to_string(), + generated: right.to_string(), + }); + } + } + + println!("{utility} ({variant})"); + let handwritten_ms = measure(|index| { + std::hint::black_box(handwritten(inputs[index % inputs.len()])); + }); + println!(" handwritten {handwritten_ms:.1} ms"); + let generated_ms = measure(|index| { + std::hint::black_box(generated(inputs[index % inputs.len()])); + }); + println!(" generated {generated_ms:.1} ms"); + + rows.push(Row { utility, variant, handwritten_ms, generated_ms }); +} + +/// Same as `compare`, but for `f64` inputs producing `String` outputs -- `formatCurrency`'s shape. +fn compare_currency( + rows: &mut Vec, + disagreements: &mut Vec, + utility: &'static str, + variant: &'static str, + inputs: &[f64], + handwritten: impl Fn(f64) -> String, + generated: impl Fn(f64) -> String, +) { + for &input in inputs { + let left = handwritten(input); + let right = generated(input); + if left != right { + disagreements.push(Disagreement { + utility, + variant, + input: input.to_string(), + handwritten: left.clone(), + generated: right.clone(), + }); + } + } + + println!("{utility} ({variant})"); + let handwritten_ms = measure(|index| { + std::hint::black_box(handwritten(inputs[index % inputs.len()])); + }); + println!(" handwritten {handwritten_ms:.1} ms"); + let generated_ms = measure(|index| { + std::hint::black_box(generated(inputs[index % inputs.len()])); + }); + println!(" generated {generated_ms:.1} ms"); + + rows.push(Row { utility, variant, handwritten_ms, generated_ms }); +} + +/// The README's generator agreement rule: `generateCpf`/`generateCnpj` draw at random, so there is +/// nothing to compare for equality. Instead, every value either side produces must validate under +/// BOTH validators -- its own port's and the generated core's -- before either side is timed. +fn check_generator_agreement( + disagreements: &mut Vec, + utility: &'static str, + variant: &'static str, + handwritten_generate: impl Fn() -> String, + handwritten_is_valid: impl Fn(&str) -> bool, + generated_generate: impl Fn() -> String, + generated_is_valid: impl Fn(&str) -> bool, +) { + for _ in 0..GENERATE_SAMPLES { + let from_handwritten = handwritten_generate(); + if !handwritten_is_valid(&from_handwritten) { + disagreements.push(Disagreement { + utility, + variant, + input: from_handwritten.clone(), + handwritten: "rejected by its own port's validator".to_owned(), + generated: "n/a".to_owned(), + }); + } + if !generated_is_valid(&from_handwritten) { + disagreements.push(Disagreement { + utility, + variant, + input: from_handwritten, + handwritten: "valid (own validator)".to_owned(), + generated: "rejected by the generated core's validator".to_owned(), + }); + } + + let from_generated = generated_generate(); + if !generated_is_valid(&from_generated) { + disagreements.push(Disagreement { + utility, + variant, + input: from_generated.clone(), + handwritten: "n/a".to_owned(), + generated: "rejected by the generated core's own validator".to_owned(), + }); + } + if !handwritten_is_valid(&from_generated) { + disagreements.push(Disagreement { + utility, + variant, + input: from_generated, + handwritten: "rejected by its own port's validator".to_owned(), + generated: "valid (own validator)".to_owned(), + }); + } + } +} + +fn compare_generate( + rows: &mut Vec, + utility: &'static str, + handwritten_generate: impl Fn() -> String, + generated_generate: impl Fn() -> String, +) { + let variant = "generate"; + println!("{utility} ({variant})"); + let handwritten_ms = measure(|_| { + std::hint::black_box(handwritten_generate()); + }); + println!(" handwritten {handwritten_ms:.1} ms"); + let generated_ms = measure(|_| { + std::hint::black_box(generated_generate()); + }); + println!(" generated {generated_ms:.1} ms"); + rows.push(Row { utility, variant, handwritten_ms, generated_ms }); +} + +/// Minimal JSON writing, so the harness needs no dependency of its own beyond the two it compares. +fn quote(value: &str) -> String { + let escaped: String = value + .chars() + .flat_map(|c| match c { + '"' => vec!['\\', '"'], + '\\' => vec!['\\', '\\'], + other => vec![other], + }) + .collect(); + format!("\"{escaped}\"") +} + +fn main() { + let mut rows = Vec::new(); + let mut disagreements = Vec::new(); + + compare( + &mut rows, + &mut disagreements, + "isValidCpf", + "normalized", + &CPFS, + brazilian_utils::cpf::is_valid, + |value| coreout::is_valid_cpf(value), + ); + compare( + &mut rows, + &mut disagreements, + "isValidCnpj", + "normalized", + &CNPJS, + brazilian_utils::cnpj::is_valid, + |value| coreout::is_valid_cnpj(value, "1"), + ); + + compare_currency( + &mut rows, + &mut disagreements, + "formatCurrency", + "normalized", + &CURRENCY_VALUES, + |value| brazilian_utils::currency::format_currency(value).unwrap(), + |value| coreout::format_currency(to_cents(value), true), + ); + + let env = BenchCapabilities::new(); // built once, like a real caller would, then reused + check_generator_agreement( + &mut disagreements, + "generateCpf", + "generate", + brazilian_utils::cpf::generate, + brazilian_utils::cpf::is_valid, + || coreout::generate_cpf(&env), + |value| coreout::is_valid_cpf(value), + ); + compare_generate(&mut rows, "generateCpf", brazilian_utils::cpf::generate, || { + coreout::generate_cpf(&env) + }); + + // brazilian_utils::cnpj::generate(None) defaults to branch=1 (fixed), not a random branch like + // the generated core -- irrelevant here, since only validity is being checked; see the README. + check_generator_agreement( + &mut disagreements, + "generateCnpj", + "generate", + || brazilian_utils::cnpj::generate(None), + brazilian_utils::cnpj::is_valid, + || coreout::generate_cnpj(&env), + |value| coreout::is_valid_cnpj(value, "1"), + ); + compare_generate(&mut rows, "generateCnpj", || brazilian_utils::cnpj::generate(None), || { + coreout::generate_cnpj(&env) + }); + + let row_json: Vec = rows + .iter() + .map(|row| { + format!( + "{{\"utility\":{},\"variant\":{},\"handwrittenMs\":{:.4},\"generatedMs\":{:.4},\"iterations\":{}}}", + quote(row.utility), + quote(row.variant), + row.handwritten_ms, + row.generated_ms, + ITERATIONS + ) + }) + .collect(); + let disagreement_json: Vec = disagreements + .iter() + .map(|item| { + format!( + "{{\"utility\":{},\"variant\":{},\"input\":{},\"handwritten\":{},\"generated\":{}}}", + quote(item.utility), + quote(item.variant), + quote(&item.input), + quote(&item.handwritten), + quote(&item.generated) + ) + }) + .collect(); + + // (utility, reason) pairs, matching the {utility, reason} object shape every other language's + // harness emits, so run.mjs's generic "Comparisons left out" printer renders these the same way. + let skipped: [(&str, &str); 4] = [ + ("formatCnpj", "brazilian_utils::cnpj::format_cnpj validates the checksum and answers None for a bad one; the generated format_cnpj never validates. Different contracts, not a fair timing comparison."), + ("isValidCnpj", "compared at version \"1\" (numeric) only: the handwritten crate has no alphanumeric CNPJ support."), + ("getHolidays", "brazilian_utils::date_utils has no getHolidays, only is_holiday(NaiveDate, Option<&str>), a single-day boolean check, not a function returning a year's list. No comparable counterpart."), + ("isBusinessDay", "brazilian_utils has no isBusinessDay or business-day/weekend concept at all -- only is_holiday, which does not consider weekends. No comparable counterpart."), + ]; + let skipped_json: Vec = skipped + .iter() + .map(|(utility, reason)| format!("{{\"utility\":{},\"reason\":{}}}", quote(utility), quote(reason))) + .collect(); + + println!( + "BENCH_JSON {{\"language\":\"rust\",\"toolchain\":{{\"rustc\":{}}},\"rows\":[{}],\"disagreements\":[{}],\"skipped\":[{}]}}", + quote(env!("BENCH_RUSTC_VERSION")), + row_json.join(","), + disagreement_json.join(","), + skipped_json.join(",") + ); +} + diff --git a/core/bench/typescript.ts b/core/bench/typescript.ts new file mode 100644 index 000000000..265e8cbcb --- /dev/null +++ b/core/bench/typescript.ts @@ -0,0 +1,381 @@ +/** + * Benchmarks the generated TypeScript against the handwritten package it replaces. + * + * Both the handwritten `src/*` functions and the generated `core/out/typescript/*` functions take + * a value "as written" (masked or not) and do their own normalization internally, so a single + * variant per utility is a fair, apples-to-apples comparison — there is no normalization work one + * side does that the other skips. This mirrors `conformance/bench.ts`, restructured to also emit + * a machine-readable result line that `run.mjs` folds into the combined table. + * + * Convention shared by every language harness in this directory: 20,000-iteration warm-up, + * 200,000 timed iterations, same process, same inputs, 1.5x budget, ratio reported as + * generated / handwritten. + */ + +import { isValidCpf as handwrittenCpf } from "../../src/is-valid-cpf/is-valid-cpf.ts"; +import { isValidCnpj as handwrittenCnpj } from "../../src/is-valid-cnpj/is-valid-cnpj.ts"; +import { formatCnpj as handwrittenFormatCnpj } from "../../src/format-cnpj/format-cnpj.ts"; +import { formatCurrency as handwrittenFormatCurrency } from "../../src/format-currency/format-currency.ts"; +import { getHolidays as handwrittenGetHolidays } from "../../src/get-holidays/get-holidays.ts"; +import { isBusinessDay as handwrittenIsBusinessDay } from "../../src/is-business-day/is-business-day.ts"; +import { generateCpf as handwrittenGenerateCpf } from "../../src/generate-cpf/generate-cpf.ts"; +import { generateCnpj as handwrittenGenerateCnpj } from "../../src/generate-cnpj/generate-cnpj.ts"; +import { isValidCpf as generatedCpf } from "../out/typescript/is-valid-cpf.ts"; +import { isValidCnpj as generatedCnpj } from "../out/typescript/is-valid-cnpj.ts"; +import { formatCnpj as generatedFormatCnpj } from "../out/typescript/format-cnpj.ts"; +import { formatCurrency as generatedFormatCurrency } from "../out/typescript/format-currency.ts"; +import { getHolidays as generatedGetHolidays, type Holiday as GeneratedHoliday } from "../out/typescript/get-holidays.ts"; +import { isBusinessDay as generatedIsBusinessDay } from "../out/typescript/is-business-day.ts"; +import { generateCpf as generatedGenerateCpf } from "../out/typescript/generate-cpf.ts"; +import { generateCnpj as generatedGenerateCnpj } from "../out/typescript/generate-cnpj.ts"; +// `std/date`'s own conversion rather than `lib/civil`'s `civilDate` wrapper: `civilDate` exists to +// name an unreachable fallback for a date the caller has not proven valid, and every call to it in +// the core passes a constant month, so it specializes away (ADR 0004) and has no single name in +// the output to import. `daysFromCivil` is the std function underneath it and stays put. +import { daysFromCivil } from "../out/typescript/std/date.ts"; + +const BUDGET = 1.5; +const WARMUP = 20_000; +const ITERATIONS = 200_000; + +const CPFS = ["123.456.789-09", "12345678909", "00000000000", "529.982.247-25", "abc"]; +const CNPJS = ["12.345.678/0001-95", "12345678000195", "00000000000000", "Q0SLFMBD7VX439"]; + +// Raw doubles, the shape `src/format-currency`'s own callers use; `toCents` turns each into the +// scaled integer (Decimal<2>) the generated core requires. None of these are round-trip-ambiguous +// doubles (no 1.005-style case), so simple rounding is enough -- the exact round-half-away-from- +// zero-on-the-shortest-decimal rule `conformance/cases.ts`'s `toScaled` implements is not needed +// here. Includes a negative value on purpose: see the README's "formatCurrency" honesty note for +// what that turns up. +const CURRENCY_VALUES = [0, 1234.56, -1234.56, 0.5, 999999.99, 10]; +const toCents = (value: number): number => Math.round(value * 100); + +// Years chosen to exercise the interesting cases: 2000 is the year `docs/contracts.md` calls out +// for the Tiradentes/Sexta-feira Santa stable-sort tie (both fall on 21 April), 2099 is the top of +// the supported range, 1987 has no Consciência Negra entry (added nationally only from 2024). +const HOLIDAY_YEARS = [2024, 2000, 2023, 1987, 2099]; + +// (year, month, day) tuples, not raw epoch-day integers or `Date`s, so the same triple can be +// turned into whichever shape each side's API wants -- a `Date` for the handwritten side, a +// `daysFromCivil` (epoch-day integer) for the generated core, per `docs/contracts.md`'s "local +// calendar day" convention. +const BUSINESS_DAY_CASES: readonly { year: number; month: number; day: number }[] = [ + { year: 2024, month: 1, day: 2 }, // ordinary Tuesday + { year: 2024, month: 1, day: 1 }, // Ano novo (national) + { year: 2024, month: 1, day: 6 }, // Saturday + { year: 2024, month: 2, day: 13 }, // Carnaval (terça-feira), optional + { year: 2024, month: 11, day: 20 }, // Consciência Negra, national since 2024 +]; + +type Row = { + utility: string; + variant: string; + handwrittenMs: number; + generatedMs: number; + iterations: number; +}; + +type Disagreement = { utility: string; variant: string; input: unknown; handwritten: unknown; generated: unknown }; + +const rows: Row[] = []; +const disagreements: Disagreement[] = []; + +function checkAgreement( + utility: string, + variant: string, + inputs: readonly string[], + handwritten: (input: string) => unknown, + generated: (input: string) => unknown, +): void { + for (const input of inputs) { + const a = handwritten(input); + const b = generated(input); + if (a !== b) disagreements.push({ utility, variant, input, handwritten: a, generated: b }); + } +} + +function measure(run: () => void): number { + for (let index = 0; index < WARMUP; index++) run(); + const started = process.hrtime.bigint(); + for (let index = 0; index < ITERATIONS; index++) run(); + return Number(process.hrtime.bigint() - started) / 1e6; +} + +function compare( + utility: string, + variant: string, + inputs: readonly string[], + handwritten: (input: string) => unknown, + generated: (input: string) => unknown, +): void { + checkAgreement(utility, variant, inputs, handwritten, generated); + + process.stdout.write(`${utility} (${variant})\n`); + let cursor = 0; + const handwrittenMs = measure(() => { + handwritten(inputs[cursor++ % inputs.length]!); + }); + process.stdout.write(` handwritten ${handwrittenMs.toFixed(1)} ms\n`); + cursor = 0; + const generatedMs = measure(() => { + generated(inputs[cursor++ % inputs.length]!); + }); + process.stdout.write(` generated ${generatedMs.toFixed(1)} ms\n`); + + rows.push({ utility, variant, handwrittenMs, generatedMs, iterations: ITERATIONS }); +} + +compare( + "isValidCpf", + "full-pipeline", + CPFS, + (input) => handwrittenCpf(input), + (input) => generatedCpf(input), +); + +compare( + "isValidCnpj", + "full-pipeline", + CNPJS, + (input) => handwrittenCnpj(input, { version: 2 }), + (input) => generatedCnpj(input, "2"), +); + +compare( + "formatCnpj", + "full-pipeline", + CNPJS, + (input) => handwrittenFormatCnpj(input, { pad: true }), + (input) => generatedFormatCnpj(input, { pad: true, version: "1", obfuscate: false }), +); + +// formatCurrency has no "full-pipeline" shape to compare at all: the generated core's contract +// (docs/contracts.md) always takes an already-scaled Decimal<2>, never a raw double -- scaling is +// DX work, done once outside the core, not something the generated side can be asked to redo. So +// this is "normalized" for the same reason Python's CPF rows are: both sides receive the value +// pre-processed into the shape their own API expects, which happens to hand the generated side +// less work than a caller starting from a raw double would. +{ + const utility = "formatCurrency"; + const variant = "normalized"; + for (const value of CURRENCY_VALUES) { + const a = handwrittenFormatCurrency(value, { symbol: true }); + const b = generatedFormatCurrency(toCents(value), true); + if (a !== b) disagreements.push({ utility, variant, input: value, handwritten: a, generated: b }); + } + + process.stdout.write(`${utility} (${variant})\n`); + let cursor = 0; + const handwrittenMs = measure(() => { + handwrittenFormatCurrency(CURRENCY_VALUES[cursor++ % CURRENCY_VALUES.length]!, { symbol: true }); + }); + process.stdout.write(` handwritten ${handwrittenMs.toFixed(1)} ms\n`); + cursor = 0; + const generatedMs = measure(() => { + generatedFormatCurrency(toCents(CURRENCY_VALUES[cursor++ % CURRENCY_VALUES.length]!), true); + }); + process.stdout.write(` generated ${generatedMs.toFixed(1)} ms\n`); + rows.push({ utility, variant, handwrittenMs, generatedMs, iterations: ITERATIONS }); +} + +// getHolidays: both sides take a plain year and do their own computation -- full-pipeline. The +// handwritten side memoizes per year (src/get-holidays/get-holidays.ts's `cache`), which the +// generated core does not do at all; see the README for what that does to this row's ratio. +{ + const utility = "getHolidays"; + const variant = "full-pipeline"; + const holidaysEqual = (year: number): { equal: boolean; handwritten: unknown; generated: unknown } => { + const handwritten = handwrittenGetHolidays(year); + const generated = generatedGetHolidays(year); + const generatedProjection = generated.map((holiday: GeneratedHoliday) => ({ + name: holiday.name, + date: holiday.date, + type: holiday.type, + })); + const handwrittenProjection = handwritten.map((holiday) => ({ + name: holiday.name, + date: daysFromCivil(holiday.date.getFullYear(), holiday.date.getMonth() + 1, holiday.date.getDate()), + type: holiday.type, + })); + const equal = JSON.stringify(handwrittenProjection) === JSON.stringify(generatedProjection); + return { equal, handwritten: handwrittenProjection, generated: generatedProjection }; + }; + for (const year of HOLIDAY_YEARS) { + const { equal, handwritten, generated } = holidaysEqual(year); + if (!equal) disagreements.push({ utility, variant, input: year, handwritten, generated }); + } + + process.stdout.write(`${utility} (${variant})\n`); + let cursor = 0; + const handwrittenMs = measure(() => { + handwrittenGetHolidays(HOLIDAY_YEARS[cursor++ % HOLIDAY_YEARS.length]!); + }); + process.stdout.write(` handwritten ${handwrittenMs.toFixed(1)} ms\n`); + cursor = 0; + const generatedMs = measure(() => { + generatedGetHolidays(HOLIDAY_YEARS[cursor++ % HOLIDAY_YEARS.length]!); + }); + process.stdout.write(` generated ${generatedMs.toFixed(1)} ms\n`); + rows.push({ utility, variant, handwrittenMs, generatedMs, iterations: ITERATIONS }); +} + +// isBusinessDay: full-pipeline. Both sides receive the same (year, month, day) local calendar day, +// each turned into the shape its own API wants -- a `Date` built from local components for the +// handwritten side, a `daysFromCivil` epoch-day integer for the generated core. +{ + const utility = "isBusinessDay"; + const variant = "full-pipeline"; + for (const { year, month, day } of BUSINESS_DAY_CASES) { + const a = handwrittenIsBusinessDay(new Date(year, month - 1, day)); + const b = generatedIsBusinessDay(daysFromCivil(year, month, day), true); + if (a !== b) { + disagreements.push({ utility, variant, input: { year, month, day }, handwritten: a, generated: b }); + } + } + + process.stdout.write(`${utility} (${variant})\n`); + let cursor = 0; + const handwrittenMs = measure(() => { + const { year, month, day } = BUSINESS_DAY_CASES[cursor++ % BUSINESS_DAY_CASES.length]!; + handwrittenIsBusinessDay(new Date(year, month - 1, day)); + }); + process.stdout.write(` handwritten ${handwrittenMs.toFixed(1)} ms\n`); + cursor = 0; + const generatedMs = measure(() => { + const { year, month, day } = BUSINESS_DAY_CASES[cursor++ % BUSINESS_DAY_CASES.length]!; + generatedIsBusinessDay(daysFromCivil(year, month, day), true); + }); + process.stdout.write(` generated ${generatedMs.toFixed(1)} ms\n`); + rows.push({ utility, variant, handwrittenMs, generatedMs, iterations: ITERATIONS }); +} + +// generateCpf / generateCnpj draw at random, so there is no fixed value to compare for equality. +// The agreement check instead of that: every value either side produces must validate under BOTH +// validators -- its own port's and the generated core's -- before either side is timed. A fast +// generator that mints invalid documents is a defect, not a pass. +const GENERATE_SAMPLES = 500; + +function checkGeneratorAgreement( + utility: string, + variant: string, + handwrittenGenerate: () => string, + handwrittenIsValid: (value: string) => boolean, + generatedGenerate: () => string, + generatedIsValid: (value: string) => boolean, +): void { + for (let index = 0; index < GENERATE_SAMPLES; index++) { + const fromHandwritten = handwrittenGenerate(); + if (!handwrittenIsValid(fromHandwritten)) { + disagreements.push({ + utility, + variant, + input: fromHandwritten, + handwritten: "rejected by its own port's validator", + generated: "n/a", + }); + } + if (!generatedIsValid(fromHandwritten)) { + disagreements.push({ + utility, + variant, + input: fromHandwritten, + handwritten: "valid (own validator)", + generated: "rejected by the generated core's validator", + }); + } + + const fromGenerated = generatedGenerate(); + if (!generatedIsValid(fromGenerated)) { + disagreements.push({ + utility, + variant, + input: fromGenerated, + handwritten: "n/a", + generated: "rejected by the generated core's own validator", + }); + } + if (!handwrittenIsValid(fromGenerated)) { + disagreements.push({ + utility, + variant, + input: fromGenerated, + handwritten: "rejected by its own port's validator", + generated: "valid (own validator)", + }); + } + } +} + +function compareGenerate( + utility: string, + handwrittenGenerate: () => string, + generatedGenerate: () => string, +): void { + const variant = "generate"; + process.stdout.write(`${utility} (${variant})\n`); + const handwrittenMs = measure(() => { + handwrittenGenerate(); + }); + process.stdout.write(` handwritten ${handwrittenMs.toFixed(1)} ms\n`); + const generatedMs = measure(() => { + generatedGenerate(); + }); + process.stdout.write(` generated ${generatedMs.toFixed(1)} ms\n`); + rows.push({ utility, variant, handwrittenMs, generatedMs, iterations: ITERATIONS }); +} + +checkGeneratorAgreement( + "generateCpf", + "generate", + () => handwrittenGenerateCpf(), + (value) => handwrittenCpf(value), + () => generatedGenerateCpf(), + (value) => generatedCpf(value), +); +compareGenerate( + "generateCpf", + () => handwrittenGenerateCpf(), + () => generatedGenerateCpf(), +); + +checkGeneratorAgreement( + "generateCnpj", + "generate", + () => handwrittenGenerateCnpj(), + (value) => handwrittenCnpj(value, { version: 1 }), + () => generatedGenerateCnpj(), + (value) => generatedCnpj(value, "1"), +); +compareGenerate( + "generateCnpj", + () => handwrittenGenerateCnpj(), + () => generatedGenerateCnpj(), +); + +process.stdout.write("\n| utility | variant | handwritten | generated | ratio | budget |\n"); +process.stdout.write("| --- | --- | --- | --- | --- | --- |\n"); +for (const row of rows) { + const ratio = row.generatedMs / row.handwrittenMs; + const ok = ratio <= BUDGET; + process.stdout.write( + `| \`${row.utility}\` | ${row.variant} | ${row.handwrittenMs.toFixed(1)} ms | ${row.generatedMs.toFixed(1)} ms | ${ratio.toFixed(2)}x | ${ok ? "within" : "OVER"} ${BUDGET}x |\n`, + ); +} + +if (disagreements.length > 0) { + process.stdout.write("\nDISAGREEMENTS:\n"); + for (const d of disagreements) { + process.stdout.write( + ` ${d.utility} (${d.variant}) input=${JSON.stringify(d.input)} handwritten=${JSON.stringify(d.handwritten)} generated=${JSON.stringify(d.generated)}\n`, + ); + } +} + +const result = { + language: "typescript", + toolchain: { node: process.version }, + rows, + disagreements, + skipped: [] as Array<{ utility: string; reason: string }>, +}; +process.stdout.write(`BENCH_JSON ${JSON.stringify(result)}\n`); diff --git a/core/conformance/bench.ts b/core/conformance/bench.ts new file mode 100644 index 000000000..c917f2ede --- /dev/null +++ b/core/conformance/bench.ts @@ -0,0 +1,88 @@ +/** + * Benchmarks the generated TypeScript against the handwritten implementation it replaces. + * + * The budget is 1.5x: generated code may be slower than the handwritten original by up to half + * again, and anything beyond that is reported. Both run in the same process, on the same inputs, + * after the same warm-up. + */ + +import { isValidCpf as handwrittenCpf } from "../../src/is-valid-cpf/is-valid-cpf.ts"; +import { isValidCnpj as handwrittenCnpj } from "../../src/is-valid-cnpj/is-valid-cnpj.ts"; +import { formatCnpj as handwrittenFormatCnpj } from "../../src/format-cnpj/format-cnpj.ts"; +import { isValidCpf as generatedCpf } from "../out/typescript/is-valid-cpf.ts"; +import { isValidCnpj as generatedCnpj } from "../out/typescript/is-valid-cnpj.ts"; +import { formatCnpj as generatedFormatCnpj } from "../out/typescript/format-cnpj.ts"; + +const BUDGET = 1.5; +const ITERATIONS = 200_000; + +const CPFS = ["123.456.789-09", "12345678909", "00000000000", "529.982.247-25", "abc"]; +const CNPJS = ["12.345.678/0001-95", "12345678000195", "00000000000000", "Q0SLFMBD7VX439"]; + +function measure(name: string, run: () => void): number { + for (let index = 0; index < 20_000; index++) run(); + const started = process.hrtime.bigint(); + for (let index = 0; index < ITERATIONS; index++) run(); + const elapsed = Number(process.hrtime.bigint() - started) / 1e6; + process.stdout.write(` ${name.padEnd(28)} ${elapsed.toFixed(1)} ms\n`); + return elapsed; +} + +type Comparison = { name: string; handwritten: number; generated: number }; + +const comparisons: Comparison[] = []; + +function compare(name: string, handwritten: () => void, generated: () => void): void { + process.stdout.write(`${name}\n`); + comparisons.push({ + name, + handwritten: measure("handwritten", handwritten), + generated: measure("generated", generated), + }); +} + +let cursor = 0; + +compare( + "isValidCpf", + () => { + handwrittenCpf(CPFS[cursor++ % CPFS.length]!); + }, + () => { + generatedCpf(CPFS[cursor++ % CPFS.length]!); + }, +); + +compare( + "isValidCnpj", + () => { + handwrittenCnpj(CNPJS[cursor++ % CNPJS.length]!, { version: 2 }); + }, + () => { + generatedCnpj(CNPJS[cursor++ % CNPJS.length]!, "2"); + }, +); + +compare( + "formatCnpj", + () => { + handwrittenFormatCnpj(CNPJS[cursor++ % CNPJS.length]!, { pad: true }); + }, + () => { + generatedFormatCnpj(CNPJS[cursor++ % CNPJS.length]!, { pad: true, version: "1", obfuscate: false }); + }, +); + +process.stdout.write("\n| utility | handwritten | generated | ratio | budget |\n| --- | --- | --- | --- | --- |\n"); + +let failed = false; +for (const comparison of comparisons) { + const ratio = comparison.generated / comparison.handwritten; + const ok = ratio <= BUDGET; + failed ||= !ok; + process.stdout.write( + `| \`${comparison.name}\` | ${comparison.handwritten.toFixed(1)} ms | ${comparison.generated.toFixed(1)} ms | ${ratio.toFixed(2)}x | ${ok ? "within" : "OVER"} ${BUDGET}x |\n`, + ); +} + +if (failed) process.exitCode = 1; diff --git a/core/conformance/cases.ts b/core/conformance/cases.ts new file mode 100644 index 000000000..8205ec678 --- /dev/null +++ b/core/conformance/cases.ts @@ -0,0 +1,337 @@ +/** + * Conformance cases for the Brazilian Utils core. + * + * Every case is a call into the core's own contract: the values a DX hands the core after its own + * coercion. Inputs come from three places — hand written vectors that pin the documented edges, + * seeded generators over the shape of the refined types, and values recorded from the published + * npm package. + */ + +import type { Case } from "../../engine/src/conformance/differential.ts"; +import { record } from "../../engine/src/values.ts"; + +/** A small deterministic generator, so a failing case is always reproducible. */ +export function makeRandom(seed: number): () => number { + let state = seed >>> 0; + return () => { + state = (state * 1_664_525 + 1_013_904_223) >>> 0; + return state / 4_294_967_296; + }; +} + +function pick(random: () => number, items: readonly T[]): T { + return items[Math.floor(random() * items.length)]!; +} + +const pickFrom = pick; + +const MASKS = [".", "-", "/", " ", "", "", "", ""]; + +export function cpfCases(): Case[] { + const cases: Case[] = []; + const push = (value: string, label?: string) => + cases.push({ fn: "is-valid-cpf::isValidCpf", args: [value], label }); + + for (const value of [ + "123.456.789-09", + "12345678909", + "123 456 789 09", + " 12345678909", + "12345678909 ", + "00000000000", + "11111111111", + "12345678900", + "", + "abc", + "111.444.777-35", + "529.982.247-25", + "529.982.247-26", + "1234567890", + "123456789091", + "123.456.789.09", + "123 456 78909", + " 123.456.789-09", + "000.000.000-00", + ]) { + push(value, "vector"); + } + + const random = makeRandom(20_260_921); + for (let index = 0; index < 500; index++) { + let digits = ""; + for (let position = 0; position < 11; position++) digits += Math.floor(random() * 10); + push(digits, "random digits"); + push( + `${digits.slice(0, 3)}${pick(random, MASKS)}${digits.slice(3, 6)}${pick(random, MASKS)}${digits.slice(6, 9)}${pick(random, MASKS)}${digits.slice(9)}`, + "random masked", + ); + } + + const alphabet = "0123456789 .-/xA"; + for (let index = 0; index < 500; index++) { + const length = Math.floor(random() * 16); + let value = ""; + for (let position = 0; position < length; position++) value += pick(random, [...alphabet]); + push(value, "fuzz"); + } + + return cases; +} + +export function cnpjCases(): Case[] { + const cases: Case[] = []; + const push = (value: string, version: "1" | "2", label?: string) => + cases.push({ fn: "is-valid-cnpj::isValidCnpj", args: [value, version], label }); + + for (const value of [ + "12.345.678/0001-95", + "12345678000195", + "12 345 678 0001 95", + "00000000000000", + "11111111111111", + "12345678000190", + "", + "Q0.SLF.MBD/7VX4-39", + "Q0SLFMBD7VX439", + "q0slfmbd7vx439", + "Q0SLFMBD7VX430", + "46.843.485/0001-86", + "46843485000186", + "AAAAAAAAAAAA00", + "1234567800019", + "123456780001955", + ]) { + push(value, "1", "vector"); + push(value, "2", "vector"); + } + + const random = makeRandom(19_881_005); + const alphabet = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZ"; + for (let index = 0; index < 400; index++) { + let value = ""; + for (let position = 0; position < 14; position++) value += pickFrom(random, [...alphabet]); + push(value, "1", "random alphanumeric"); + push(value, "2", "random alphanumeric"); + + let digits = ""; + for (let position = 0; position < 14; position++) digits += Math.floor(random() * 10); + push(digits, "1", "random numeric"); + push(digits, "2", "random numeric"); + push( + `${digits.slice(0, 2)}.${digits.slice(2, 5)}.${digits.slice(5, 8)}/${digits.slice(8, 12)}-${digits.slice(12)}`, + "1", + "random masked", + ); + } + + return cases; +} + +export function formatCnpjCases(): Case[] { + const cases: Case[] = []; + const push = (value: string, pad: boolean, version: "1" | "2", obfuscate: boolean) => + cases.push({ + fn: "format-cnpj::formatCnpj", + args: [value, record("FormatCnpjOptions", { pad, version, obfuscate })], + label: "formatCnpj", + }); + + const values = [ + "", + "4", + "46", + "468", + "4684", + "46843", + "468434", + "4684348", + "46843485", + "468434850", + "4684348500", + "46843485000", + "468434850001", + "4684348500018", + "46843485000186", + "468434850001866", + "12.345.678/0001-95", + "q0SLFMBD7VX439", + "Q0.SLF.MBD/7VX4-39", + "abc", + ]; + + for (const value of values) { + for (const pad of [false, true]) { + for (const version of ["1", "2"] as const) { + for (const obfuscate of [false, true]) push(value, pad, version, obfuscate); + } + } + } + + return cases; +} + +/** Days since 1970-01-01 for a local calendar day, the conversion the DX owns. */ +export function epochDays(year: number, month: number, day: number): bigint { + return BigInt(Math.floor(Date.UTC(year, month - 1, day) / 86_400_000)); +} + +export function holidayCases(): Case[] { + const cases: Case[] = []; + for (const year of [1900, 1999, 2000, 2020, 2023, 2024, 2025, 2026, 2030, 2099]) { + cases.push({ fn: "get-holidays::getHolidays", args: [BigInt(year)], label: "holidays" }); + } + const random = makeRandom(7_2024); + for (let index = 0; index < 60; index++) { + cases.push({ + fn: "get-holidays::getHolidays", + args: [BigInt(1900 + Math.floor(random() * 200))], + label: "holidays", + }); + } + return cases; +} + +export function businessDayCases(): Case[] { + const cases: Case[] = []; + const push = (year: number, month: number, day: number, includeOptional: boolean) => + cases.push({ + fn: "is-business-day::isBusinessDay", + args: [epochDays(year, month, day), includeOptional], + label: `${year}-${month}-${day}`, + }); + + for (const [year, month, day] of [ + [2024, 1, 2], + [2024, 1, 1], + [2024, 1, 6], + [2024, 1, 7], + [2024, 2, 13], + [2024, 3, 29], + [2024, 5, 30], + [2024, 11, 20], + [2023, 11, 20], + [2025, 12, 25], + [2026, 4, 21], + [1900, 1, 1], + [2099, 12, 31], + ]) { + push(year!, month!, day!, true); + push(year!, month!, day!, false); + } + + const random = makeRandom(31_012_024); + for (let index = 0; index < 400; index++) { + const year = 1900 + Math.floor(random() * 200); + const month = 1 + Math.floor(random() * 12); + const day = 1 + Math.floor(random() * 28); + push(year, month, day, random() > 0.5); + } + + return cases; +} + +/** + * Scripted Http, shared by the reference interpreter and every generated target. + * + * A URL that is absent models a transport error, and the latency is what decides a race; the + * reference model resolves it on virtual time, the targets on real time, and both must pick the + * same winner. + */ +export const HTTP_FIXTURES: Record = { + "https://viacep.com.br/ws/01310100/json/": { + status: 200, + body: '{"cep":"01310-100","logradouro":"Avenida Paulista","bairro":"Bela Vista","localidade":"S\u00e3o Paulo","uf":"SP"}', + latencyMillis: 60, + }, + "https://brasilapi.com.br/api/cep/v1/01310100": { + status: 200, + body: '{"cep":"01310100","state":"SP","city":"S\u00e3o Paulo","neighborhood":"Bela Vista","street":"Avenida Paulista"}', + latencyMillis: 20, + }, + // Only BrasilAPI knows this one, and it answers slowly. + "https://brasilapi.com.br/api/cep/v1/30130010": { + status: 200, + body: '{"cep":"30130010","state":"MG","city":"Belo Horizonte","neighborhood":"Centro","street":"Avenida Afonso Pena"}', + latencyMillis: 40, + }, + "https://viacep.com.br/ws/99999999/json/": { status: 200, body: '{"erro":true}', latencyMillis: 10 }, + "https://brasilapi.com.br/api/cep/v1/99999999": { status: 404, body: '{"message":"not found"}', latencyMillis: 10 }, + // Neither service answers: every attempt is a transport error, so the retry policy runs. +}; + +export function cepCases(): Case[] { + const cases: Case[] = []; + for (const cep of ["01310100", "30130010", "99999999", "12345678", "0131010", "abcdefgh", ""]) { + cases.push({ fn: "get-address-info-by-cep::getAddressInfoByCep", args: [cep], label: "cep" }); + } + return cases; +} + +/** + * The DX conversion the published package performs before formatting. + * + * `Intl.NumberFormat` rounds the *shortest round-trip decimal* representation of the double, not + * its exact binary value, which is why 1.005 formats as "1,01" while `toFixed(2)` answers "1.00". + * The core takes the exact amount that conversion produces. + */ +export function toScaled(value: number, precision: number): bigint { + const text = String(value); + const negative = text.startsWith("-"); + const digits = negative ? text.slice(1) : text; + const [whole = "0", fraction = ""] = digits.split("."); + const padded = `${fraction}${"0".repeat(precision + 1)}`; + const kept = BigInt(`${whole}${padded.slice(0, precision)}`); + const next = Number(padded[precision]); + const rounded = next >= 5 ? kept + 1n : kept; + return negative ? -rounded : rounded; +} + +export function currencyCases(): Case[] { + const cases: Case[] = []; + const values = [ + 0, 1, -1, 0.5, 1.005, 2.675, 10.5, -10.5, 1234.56, -1234.56, 999999.99, 1_000_000, 123, 0.01, + -0.01, 0.001, 1e9, 12345678.9, -0.5, + ]; + + for (const value of values) { + for (const symbol of [false, true]) { + cases.push({ + fn: "format-currency::formatCurrency", + args: [toScaled(value, 2), symbol], + label: `${value}`, + // The reference call needs the original double, which the scaled amount no longer carries. + reference: [value, symbol], + } as Case); + } + } + + return cases; +} + +/** + * `generateCpf` and `generateCnpj` call `Math.random()` in the published package, so there is no + * fixed value to compare against; what the reference interpreter and the three targets have to + * agree on is the *draw itself*, under the one default seed every case in this suite runs with. + * `run.ts` turns each of these into a follow-up `isValidCpf`/`isValidCnpj` case, built from + * whatever the reference interpreter actually drew, so a target that matches on the wrong value + * (the same bug reproduced identically) still gets caught by the check digits. + */ +export function generateCases(): Case[] { + return [ + { fn: "generate-cpf::generateCpf", args: [], label: "generateCpf" }, + { fn: "generate-cnpj::generateCnpj", args: [], label: "generateCnpj" }, + ]; +} + +export function allCases(): Case[] { + return [ + ...cpfCases(), + ...cnpjCases(), + ...formatCnpjCases(), + ...holidayCases(), + ...businessDayCases(), + ...cepCases(), + ...currencyCases(), + ...generateCases(), + ]; +} diff --git a/core/conformance/run-source.ts b/core/conformance/run-source.ts new file mode 100644 index 000000000..a4803d9b9 --- /dev/null +++ b/core/conformance/run-source.ts @@ -0,0 +1,210 @@ +/** + * Runs `core/source` itself, in Node, against the conformance vectors — no engine involved: no + * compilation, no generation, just the migrated TypeScript executing. `verify.ts`'s "check" and + * "generate" steps only prove the source compiles; this is the test of the harder claim, that it + * actually runs and computes the right thing, so without this step the claim is not made. + * + * Each covered utility is called directly and compared against the published npm package + * (`src/`), the same ground truth `conformance/run.ts` compares the reference interpreter + * against. Not every utility can run yet: `format-currency` needs `dec.*` (no exact decimal in + * JavaScript), `generate-cpf`/`generate-cnpj` need `random.nextU32` (an effect with no unbiased + * `Math.random` equivalent), `get-holidays`/`is-business-day` need `date.*` (`new Date` is + * refused), and `get-address-info-by-cep` needs `http.request`/`task.race` (no shared meaning with + * `fetch`'s `Promise`) — see core/docs/idiomatic-migration.md for the full reasoning on + * each. `results` below is the exact, closed set this step exercises; a utility is added here only + * once at least one of its cases can run. + * + * `isValidCpf` and `formatCnpj` run end to end for every case: both were blocked, before the + * checked accessors (`str.charAtOpt`, `str.codeAtOpt`, `seq.at`) gained ordinary spellings, on a + * checked positional read with no provable bound (`formatWithPattern`'s scan of the format + * pattern, and `isValidCpf`'s own digit scan) — now `pattern[index] ?? ""` and its kin. + * + * `isValidCnpj` is partial, but for a different reason than it used to be: the alphanumeric + * (version `"2"`) letter-detection helper (`hasLetter`, in `lib/cnpj.ts`) is fully migrated now — + * `value[index]?.charCodeAt(0) ?? 0` is the ordinary spelling of the checked *numeric* accessor + * `str.codeAtOpt` this migration adds — so it no longer blocks anything. What still does is a + * separate, unrelated gap `is-valid-cnpj.ts` documents in place: once a value is confirmed + * alphanumeric and the right length, the checker validates its *unsanitized, trimmed* shape with + * `CNPJ_FORMAT.test(str.asciiUpper(trimmed))`, and `trimmed` (`cnpj.trim()` on the raw external + * input) carries no ASCII proof, so `.toUpperCase()` — the ordinary spelling this migration also + * adds for `str.asciiUpper`, but only on a proven-ASCII argument — is not available to it. Every + * case is still attempted rather than pre-filtered, so a case that reaches that line is confirmed + * to fail with exactly that gap, not silently dropped — a change that fixes or breaks it is + * caught either way. + */ + +import { isValidCpf as sourceIsValidCpf } from "../source/is-valid-cpf"; +import { isValidCnpj as sourceIsValidCnpj } from "../source/is-valid-cnpj"; +import { formatCnpj as sourceFormatCnpj } from "../source/format-cnpj"; + +import { isValidCpf as referenceIsValidCpf } from "../../src/is-valid-cpf/is-valid-cpf"; +import { isValidCnpj as referenceIsValidCnpj } from "../../src/is-valid-cnpj/is-valid-cnpj"; +import { formatCnpj as referenceFormatCnpj } from "../../src/format-cnpj/format-cnpj"; + +import { cnpjCases, cpfCases, formatCnpjCases } from "./cases"; + +/** + * The `ReferenceError` an unmigrated ambient identifier (`str`, `seq`, `re`, `int`, `dec`, + * `date`, `random`, `task`, `clock`, `http`) throws when the real module actually executes that + * line — the signature of the one documented, known gap, not a stand-in for "anything failed". + */ +function isAmbientGap(error: unknown): boolean { + return ( + error instanceof ReferenceError && + /^(str|seq|re|int|dec|date|random|task|clock|http) is not defined$/.test(error.message) + ); +} + +type Case = { readonly fn: string; readonly args: readonly unknown[]; readonly label?: string }; + +type Coverage = { + readonly name: string; + total: number; + matched: number; + blocked: number; + readonly failures: string[]; +}; + +/** + * Runs one utility's cases through the migrated source and the published package, and reports + * three outcomes per case: matched (both agree), blocked (the source hit the one documented + * ambient gap), or a failure (anything else — a real mismatch, or an unexpected throw). + */ +function run( + name: string, + cases: readonly Case[], + call: (args: readonly unknown[]) => unknown, + reference: (args: readonly unknown[]) => unknown, +): Coverage { + const coverage: Coverage = { name, total: cases.length, matched: 0, blocked: 0, failures: [] }; + + for (const testCase of cases) { + let actual: unknown; + try { + actual = call(testCase.args); + } catch (error) { + if (isAmbientGap(error)) { + coverage.blocked += 1; + } else { + coverage.failures.push(`${name}(${JSON.stringify(testCase.args)}) threw ${String(error)}`); + } + continue; + } + + const expected = reference(testCase.args); + if (JSON.stringify(actual) === JSON.stringify(expected)) { + coverage.matched += 1; + } else { + coverage.failures.push( + `${name}(${JSON.stringify(testCase.args)}) expected ${JSON.stringify(expected)}, got ${JSON.stringify(actual)}`, + ); + } + } + + return coverage; +} + +const results: Coverage[] = [ + run( + "is-valid-cpf::isValidCpf", + cpfCases(), + (args) => sourceIsValidCpf(args[0] as string), + (args) => referenceIsValidCpf(args[0] as string), + ), + run( + "is-valid-cnpj::isValidCnpj", + cnpjCases(), + (args) => sourceIsValidCnpj(args[0] as string, args[1] as "1" | "2"), + (args) => referenceIsValidCnpj(args[0] as string, { version: args[1] === "2" ? 2 : 1 }), + ), + run( + "format-cnpj::formatCnpj", + formatCnpjCases(), + (args) => { + // `cases.ts` builds the options as an engine `record` Value (`{ fields: {...} }`), the + // same shape `conformance/run.ts` unwraps for its own reference call — `formatCnpj`'s + // contract wants the plain option record a DX would have already normalized. + const fields = (args[1] as { fields: Record }).fields; + return sourceFormatCnpj(args[0] as string, { + pad: fields["pad"] as boolean, + version: fields["version"] === "2" ? "2" : "1", + obfuscate: fields["obfuscate"] as boolean, + }); + }, + (args) => { + const fields = (args[1] as { fields: Record }).fields; + // The published package's own `version` is numeric (1 | 2), unlike the core's "1" | "2". + return referenceFormatCnpj(args[0] as string, { + pad: fields["pad"] as boolean, + version: fields["version"] === "2" ? 2 : 1, + obfuscate: fields["obfuscate"] as boolean, + }); + }, + ), +]; + +/** Utilities this step deliberately does not attempt, and why — see the file header for detail. */ +const NOT_COVERED = [ + "format-currency::formatCurrency (dec.*, no exact decimal in JavaScript)", + "generate-cpf::generateCpf (random.nextU32, an effect)", + "generate-cnpj::generateCnpj (random.nextU32, an effect)", + "get-holidays::getHolidays (date.*, new Date has no proleptic-Gregorian equivalent)", + "is-business-day::isBusinessDay (date.*, same as get-holidays)", + "get-address-info-by-cep::getAddressInfoByCep (http.request/task.race, effects; also needs a live network to compare)", +]; + +let failed = false; + +for (const coverage of results) { + const status = coverage.failures.length === 0 ? "ok" : "FAILED"; + process.stdout.write( + `${status.padEnd(7)} ${coverage.name}: ${coverage.matched} matched, ${coverage.blocked} blocked (known gap), ` + + `${coverage.total} total\n`, + ); + for (const failure of coverage.failures.slice(0, 5)) { + process.stdout.write(` ${failure}\n`); + } + if (coverage.failures.length > 5) { + process.stdout.write(` … and ${coverage.failures.length - 5} more\n`); + } + if (coverage.failures.length > 0) failed = true; +} + +process.stdout.write(`skipped ${NOT_COVERED.length} utilities, blocked on an intrinsic with no ordinary spelling yet:\n`); +for (const entry of NOT_COVERED) { + process.stdout.write(` ${entry}\n`); +} + +// `isValidCpf` and `formatCnpj` are fully migrated: every case must run for real, none may fall +// back to the known-gap path, or this step would be quietly certifying less than it claims. +for (const name of ["is-valid-cpf::isValidCpf", "format-cnpj::formatCnpj"]) { + const coverage = results.find((entry) => entry.name === name)!; + if (coverage.blocked !== 0) { + process.stdout.write(`FAILED ${coverage.name}: expected to run fully, but ${coverage.blocked} case(s) hit an ambient gap\n`); + failed = true; + } + if (coverage.total === 0) { + process.stdout.write(`FAILED ${coverage.name}: 0 cases — the vector set shrank to nothing\n`); + failed = true; + } +} + +// `isValidCnpj` is only partially migrated (see the file header): every case must either match +// the reference or hit exactly the documented gap, and at least one case of each must occur, so +// neither half of the claim ("this runs" / "this specific part doesn't yet") can drift unnoticed. +{ + const coverage = results.find((entry) => entry.name === "is-valid-cnpj::isValidCnpj")!; + if (coverage.matched === 0) { + process.stdout.write(`FAILED ${coverage.name}: 0 cases matched — the runnable part stopped running\n`); + failed = true; + } + if (coverage.blocked === 0) { + process.stdout.write( + `FAILED ${coverage.name}: 0 cases hit the known gap — either it was fixed (update this step and ` + + "core/docs/idiomatic-migration.md to say so) or the vectors stopped exercising it\n", + ); + failed = true; + } +} + +if (failed) process.exitCode = 1; diff --git a/core/conformance/run.ts b/core/conformance/run.ts new file mode 100644 index 000000000..8935c802b --- /dev/null +++ b/core/conformance/run.ts @@ -0,0 +1,227 @@ +/** + * The differential runner. + * + * Compiles the project, generates every target in both idiom modes, and compares the reference + * interpreter, the published npm package and each generated target on the same cases. + */ + +import { existsSync, writeFileSync } from "node:fs"; +import { resolve } from "node:path"; +import { compileProject } from "../../engine/src/api.ts"; +import { compare, runInterpreter, runTarget } from "../../engine/src/conformance/differential.ts"; +import type { Case, Divergence, Outcome } from "../../engine/src/conformance/differential.ts"; +import { HTTP_FIXTURES, allCases } from "./cases.ts"; +import { formatCurrency } from "../../src/format-currency/format-currency.ts"; +import { isValidCpf } from "../../src/is-valid-cpf/is-valid-cpf.ts"; +import { isValidCnpj } from "../../src/is-valid-cnpj/is-valid-cnpj.ts"; +import { formatCnpj } from "../../src/format-cnpj/format-cnpj.ts"; +import type { Value } from "../../engine/src/values.ts"; +import { getHolidays } from "../../src/get-holidays/get-holidays.ts"; +import { isBusinessDay } from "../../src/is-business-day/is-business-day.ts"; + +const ROOT = resolve(import.meta.dirname, ".."); + +/** What the published package answers, which is the behavior the core must reproduce. */ +const REFERENCE: Record unknown> = { + "is-valid-cpf::isValidCpf": (args) => isValidCpf(args[0] as string), + // The DX maps its own options onto the core's contract: a `version` that is not 2 is read as + // the numeric format, exactly as the published package documents. + "is-valid-cnpj::isValidCnpj": (args) => + isValidCnpj(args[0] as string, { version: args[1] === "2" ? 2 : 1 }), + // The DX turns a host `Date` into a civil date and back; the core never sees a zone. + "get-holidays::getHolidays": (args) => + getHolidays({ year: Number(args[0] as bigint) }).map((holiday) => ({ + name: holiday.name, + date: Math.floor( + Date.UTC(holiday.date.getFullYear(), holiday.date.getMonth(), holiday.date.getDate()) / 86_400_000, + ), + type: holiday.type, + })), + "is-business-day::isBusinessDay": (args) => { + const days = Number(args[0] as bigint); + const utc = new Date(days * 86_400_000); + return isBusinessDay(new Date(utc.getUTCFullYear(), utc.getUTCMonth(), utc.getUTCDate()), { + includeOptional: args[1] as boolean, + }); + }, + // The core takes the exact amount; the double and the rounding rule stay on the DX side. + "format-currency::formatCurrency": (args) => + formatCurrency(args[0] as number, { symbol: args[1] as boolean }), + "format-cnpj::formatCnpj": (args) => { + const options = args[1] as { fields: Record }; + return formatCnpj(args[0] as string, { + pad: options.fields["pad"] as boolean, + version: options.fields["version"] === "2" ? 2 : 1, + obfuscate: options.fields["obfuscate"] as boolean, + }); + }, +}; + +/** + * What the published package answers, for the utilities whose behavior can be reproduced offline. + * + * `getAddressInfoByCep` is not among them: the published implementation performs real requests, so + * its contract is recorded in docs/contracts.md and checked against the reference interpreter and + * the three targets instead. + */ +function runReference(cases: readonly Case[]): { outcomes: Outcome[]; compared: number[] } { + const outcomes: Outcome[] = []; + const compared: number[] = []; + for (const [index, testCase] of cases.entries()) { + const fn = REFERENCE[testCase.fn]; + if (fn === undefined) { + outcomes.push({ ok: true, value: null }); + continue; + } + compared.push(index); + const reference = (testCase as { reference?: readonly Value[] }).reference; + outcomes.push({ ok: true, value: fn(reference ?? testCase.args) }); + } + return { outcomes, compared }; +} + +/** + * How each target's generated driver is started. + * + * `directory` is where that target's generated code lives, which is where its capability fake + * reads `fixtures.json` from; Python runs as a package, so its working directory is the parent. + */ +function runners(mode: "idiomatic" | "plain") { + const suffix = mode === "plain" ? "-plain" : ""; + return [ + { + name: `typescript${suffix}`, + command: process.execPath, + args: ["_driver.ts"], + cwd: resolve(ROOT, `out/typescript${suffix}`), + directory: resolve(ROOT, `out/typescript${suffix}`), + }, + { + name: `python${suffix}`, + command: "python3", + args: ["-m", `python${suffix}._driver`], + cwd: resolve(ROOT, "out"), + directory: resolve(ROOT, `out/python${suffix}`), + }, + { + name: `go${suffix}`, + command: "go", + args: ["run", "./cmd/driver"], + cwd: resolve(ROOT, `out/go${suffix}`), + directory: resolve(ROOT, `out/go${suffix}`), + }, + { + name: `rust${suffix}`, + command: "cargo", + // `--release`: `task.race` is real threads over a real (if fake) network call, and a + // debug build's overflow checks have nothing to catch — every range is already proven, + // same as the other three targets — so there is no reason to pay for them here. + args: ["run", "--offline", "--release", "--quiet", "--bin", "driver"], + cwd: resolve(ROOT, `out/rust${suffix}`), + directory: resolve(ROOT, `out/rust${suffix}`), + }, + ]; +} + +function main(): void { + const only = process.argv.includes("--target") ? process.argv[process.argv.indexOf("--target") + 1] : undefined; + const modes: ("idiomatic" | "plain")[] = process.argv.includes("--idiomatic-only") + ? ["idiomatic"] + : ["idiomatic", "plain"]; + + const compilation = compileProject(resolve(ROOT, "source")); + let cases = allCases(); + + // Every target reads the same scripted Http, so a race is decided by the same latencies. + for (const mode of ["idiomatic", "plain"] as const) { + for (const runner of runners(mode)) { + if (!existsSync(runner.directory)) continue; + writeFileSync( + resolve(runner.directory, "fixtures.json"), + `${JSON.stringify(HTTP_FIXTURES, null, "\t")}\n`, + ); + } + } + + let reference = runInterpreter(compilation.program, cases, { + http: (request) => { + const fixture = HTTP_FIXTURES[request.url]; + return fixture === undefined + ? undefined + : { status: fixture.status, body: fixture.body, latencyMillis: fixture.latencyMillis }; + }, + }); + + // `generateCpf`/`generateCnpj` have no published reference to compare against, only each + // other; this checks the reference interpreter's own draw is a document that actually + // validates, so a target that reproduces the same wrong draw in every language (matching the + // interpreter bit for bit, but on a broken value) still gets caught by the check digits. + const validityCases: Case[] = []; + cases.forEach((testCase, index) => { + const outcome = reference[index]!; + if (!outcome.ok) return; + if (testCase.fn === "generate-cpf::generateCpf") { + validityCases.push({ + fn: "is-valid-cpf::isValidCpf", + args: [outcome.value as string], + label: "generateCpf produces a valid CPF", + }); + } else if (testCase.fn === "generate-cnpj::generateCnpj") { + validityCases.push({ + fn: "is-valid-cnpj::isValidCnpj", + args: [outcome.value as string, "1"], + label: "generateCnpj produces a valid CNPJ", + }); + } + }); + if (validityCases.length > 0) { + reference = [...reference, ...runInterpreter(compilation.program, validityCases, {})]; + cases = [...cases, ...validityCases]; + } + + const { outcomes: npmOutcomes, compared } = runReference(cases); + const comparedCases = compared.map((index) => cases[index]!); + const npmDivergences = compare( + compared.map((index) => npmOutcomes[index]!), + compared.map((index) => reference[index]!), + comparedCases, + "interpreter vs npm", + ); + report("interpreter vs npm", comparedCases.length, npmDivergences); + + let failures = npmDivergences.length; + + for (const mode of modes) { + for (const runner of runners(mode)) { + if (only !== undefined && !runner.name.startsWith(only)) continue; + if (!existsSync(runner.cwd)) { + process.stdout.write(`${runner.name}: skipped, ${runner.cwd} is missing\n`); + continue; + } + const outcomes = runTarget(runner, cases); + const divergences = compare(reference, outcomes, cases, runner.name); + report(runner.name, cases.length, divergences); + failures += divergences.length; + } + } + + if (failures > 0) process.exitCode = 1; +} + +/** BigInt is not JSON, and a divergence report has to print one. */ +function show(value: unknown): string { + return JSON.stringify(value, (_key, item: unknown) => (typeof item === "bigint" ? item.toString() : item)); +} + +function report(name: string, total: number, divergences: readonly Divergence[]): void { + const matched = total - divergences.length; + process.stdout.write(`${name}: ${matched}/${total} matched\n`); + for (const divergence of divergences.slice(0, 5)) { + process.stdout.write( + ` ${divergence.case.fn}(${show(divergence.case.args)}) expected ${show(divergence.expected)}, got ${show(divergence.actual)}\n`, + ); + } + if (divergences.length > 5) process.stdout.write(` … and ${divergences.length - 5} more\n`); +} + +main(); diff --git a/core/conformance/sloppy-imports.mjs b/core/conformance/sloppy-imports.mjs new file mode 100644 index 000000000..728dcb8ad --- /dev/null +++ b/core/conformance/sloppy-imports.mjs @@ -0,0 +1,22 @@ +/** + * Lets Node import the published package's own sources, which use extensionless specifiers + * (the repository builds with a bundler). The conformance harness compares the engine against + * that source, so it has to load it exactly as written. + */ +import { registerHooks } from "node:module"; +import { existsSync } from "node:fs"; +import { fileURLToPath, pathToFileURL } from "node:url"; +import { dirname, resolve as resolvePath } from "node:path"; + +registerHooks({ + resolve(specifier, context, nextResolve) { + if (specifier.startsWith(".") && !/\.[cm]?[jt]s$/.test(specifier)) { + const parent = context.parentURL === undefined ? process.cwd() : dirname(fileURLToPath(context.parentURL)); + for (const candidate of [`${specifier}.ts`, `${specifier}/index.ts`]) { + const full = resolvePath(parent, candidate); + if (existsSync(full)) return { url: pathToFileURL(full).href, shortCircuit: true }; + } + } + return nextResolve(specifier, context); + }, +}); diff --git a/core/docs/contracts.md b/core/docs/contracts.md new file mode 100644 index 000000000..26e83ed25 --- /dev/null +++ b/core/docs/contracts.md @@ -0,0 +1,141 @@ +# Contracts + +What the published package actually does, measured rather than assumed, split into what the +**core** owes and what the **DX** owes. The core is generated; the DX stays handwritten in each +language's repository. + +Every fact here was measured against the sources in `../src` on Node 22 with ICU 78.2, and is +pinned by a case in `conformance/cases.ts`. + +--- + +## `isValidCpf` + +**Core.** The value is trimmed and matched against +`^[0-9]{3}M*[0-9]{3}M*[0-9]{3}M*[0-9]{2}$`, where `M` is the mask class. The digits are then +extracted from the *untrimmed* value, which is equivalent because the mask class contains no +digits. A value whose digits are all the same is rejected. The two check digits follow the Receita +Federal rule (weights 10…2 and 11…2, remainder below 2 meaning 0). + +**Measured details.** + +- The package's `\s` inside the mask class is JavaScript's `\s`: the 25 code points listed in + `docs/semantics.md`. The core spells them out, because `\s` means a different set in Python and + in Go. +- Mask characters are allowed *between* groups and any number of them, so `"123 456 789 09"` + is valid and `"123.456.789.09"` is valid too. +- A value with 12 digits fails the format check; a value with 10 fails it as well. + +**DX.** A non-string returns `false` without reaching the core. + +--- + +## `isValidCnpj` + +**Core.** `isValidCnpj(value: string, version: "1" | "2")`. + +- Under `"2"`, the alphanumeric path runs only when the sanitized value contains an upper-cased + ASCII letter; otherwise the numeric path runs. The alphanumeric path upper-cases the trimmed + value before the format check and does **not** apply the repeated-character rule, because the + Receita Federal manual defines no reserved values for the alphanumeric format. +- Both paths share one check digit calculation: each character contributes its code point minus + 48, which is the digit itself for `0`–`9` and the value the manual assigns to `A`–`Z` (17 to + 42). + +**Measured details.** + +- The numeric path rejects a repeated value (`"00000000000000"`), the alphanumeric path does not. +- The sanitizers read the original value, not the trimmed one. + +**DX.** `options.version` is mapped onto `"1"` or `"2"`: the published package reads *any* value +other than `2` as the numeric format, and the DX reproduces that. + +**Documented divergence.** The package upper-cases with `String#toUpperCase`, which maps some +non-ASCII scalars into ASCII (U+0131 "ı" becomes "I"). The core's case mapping is ASCII-only, so +such a value is rejected where the package would accept it. Full Unicode case mapping is out of +scope (`docs/semantics.md`, "Strings"). + +--- + +## `formatCnpj` + +**Core.** `formatCnpj(value: string, options: { pad, version, obfuscate })`, with the pattern +`00.000.000/0000-00` or, when obfuscating, `**.000.000/0000-**`. + +The pattern is read scalar by scalar: `0` copies one input scalar, `*` hides one, anything else is +a separator emitted only while the value still has scalars left. With `pad`, the value is left +padded with zeros to the number of slots the pattern has — which is why `formatCnpj("4", { pad: +true })` is `"00.000.000/0000-04"` and `formatCnpj("")` is `""`. + +**DX.** A number is converted with `String(value)`; a missing options object becomes +`{ pad: false, version: "1", obfuscate: false }`; `obfuscate` is read for truthiness, so `1` +obfuscates. + +--- + +## `getHolidays` and `isBusinessDay` + +**Core.** `getHolidays(year)` answers the national holidays of a year sorted by date, with a +**stable** sort, so two holidays on the same day keep the order they were built in — which is +observable in 2000, where Tiradentes and Sexta-feira Santa both fall on 21 April and Tiradentes +comes first. + +The build order is the one the package uses: the eight fixed holidays in statutory order, then Dia +da Consciência Negra from 2024 (Lei 14.759/2023), then Carnaval (Easter − 47), Sexta-feira Santa +(Easter − 2), Páscoa (Easter) and Corpus Christi (Easter + 60). Easter is Meeus/Jones/Butcher. + +`isBusinessDay(date, includeOptional)` answers `false` on a weekend, on any listed holiday, and +outside 1900–2099. + +**DX.** The host `Date` is interpreted as a **local** calendar day — the package reads +`getFullYear`, `getMonth` and `getDate` — and the DX converts that day into a `CivilDate`. The +core never sees a zone. `includeOptional` defaults to `true`. + +**Not in the pilot.** State holidays, the Santa Catarina next-Sunday rule, and the `stateCode` +handling (including the non-string rejection) stay with the package for now. + +--- + +## `getAddressInfoByCep` + +**Core.** The value must be exactly eight digits, or the core raises +`GetAddressInfoByCepValidationError`. ViaCEP and BrasilAPI are then queried concurrently, each GET +retried twice more at 250 ms, and the first service that answers wins. A service answers when its +payload carries a non-empty `cep` field. When neither does, the core raises +`GetAddressInfoByCepNotFoundError`. + +**Measured details.** + +- Provider URLs: `https://viacep.com.br/ws//json/` and + `https://brasilapi.com.br/api/cep/v1/`; the package's default provider list is + `["viacep", "brasilapi"]`, in that order. +- The retry policy is the package's `fetchWithRetry` default: 2 retries, 250 ms apart. +- The returned `cep` has its mask removed, which matters for ViaCEP (`"01310-100"`). + +**DX.** Sanitizing the input to digits, left padding a number to eight, and the third provider +(WideNet) stay in the DX for now. The published package performs real requests, so conformance +drives the core with scripted responses instead (`conformance/cases.ts`), and the reference +interpreter and all three targets are compared against each other on them. + +--- + +## `formatCurrency` + +**Core.** `formatCurrency(value: Decimal<2>, symbol: boolean)` formats an exact amount: `.` between +thousands, `,` before the centavos, and with `symbol` the prefix `R$` followed by **an ordinary +space**. + +**Measured details.** + +- `Intl.NumberFormat("pt-BR", { style: "currency", currency: "BRL" })` emits U+00A0 after `R$`, + and the package replaces it with U+0020 before returning. The core produces the ordinary space, + matching the package rather than CLDR. +- A negative amount puts the sign before the symbol: `-R$ 10,50`. +- Rounding belongs to the DX, and it is **not** `toFixed`. `Intl` rounds the shortest round-trip + decimal representation of the double, half away from zero, so `1.005` formats as `"1,01"` while + `(1.005).toFixed(2)` is `"1.00"`, and `2.675` formats as `"2,68"` while `toFixed` gives + `"2.67"`. `conformance/cases.ts` implements exactly that rule in `toScaled`. + +**Not in the pilot.** The `precision` option (0 to 20), string inputs read by `parseCurrency`'s +rule, and the empty string answered for a non-finite value. The first needs a run-time scale, +which `Decimal` deliberately does not have; the other two are DX coercion. diff --git a/core/docs/idiomatic-migration.md b/core/docs/idiomatic-migration.md new file mode 100644 index 000000000..355d1eb24 --- /dev/null +++ b/core/docs/idiomatic-migration.md @@ -0,0 +1,295 @@ +# Migrating `core/source` to the ordinary spelling + +`core/source` used to compile — `tsc` accepted it — without running: `str`, `re`, `seq`, `int`, +`date`, `dec`, `random` and `task` are ambient declarations (`engine/prelude/index.d.ts`) with +nothing behind them at runtime. This records what moved to the ordinary spelling the frontend now +recognizes (`engine/docs/semantics.md` §7.1), what still cannot move and why, and how the claim +"the source actually runs" is now checked rather than only asserted. + +A second pass (this update) gave the **checked accessors** — `str.charAtOpt`, `str.codeAtOpt`, +`seq.at` — an ordinary spelling. They had none before: the frontend's provability rule for +`s[i]`/`xs[i]` picked the *unchecked* form whenever the index was proven in range, which is +exactly backwards for the three call sites that needed the checked form on purpose (see "`??` +forces the checked accessor" below), and `str.codeAtOpt` had no bracket form at all to pick +between. Giving these a spelling, and migrating what it unblocks, is what moved `formatCnpj` from +not running at all to matching on every case, and `isValidCnpj` from 1,216/2,032 to 1,627/2,032. + +--- + +## What moved + +Every occurrence the recognizer supports moved, across `core/source/**`: + +| file | namespace form | ordinary form | +| --- | --- | --- | +| `lib/digits.ts` | `re.retain(DIGIT, value)` / `re.retain(ALPHANUMERIC, value)` | `value.replace(/[^0-9]/g, "")` / `value.replace(/[^0-9A-Za-z]/g, "")` | +| `lib/digits.ts` | `str.codeAt(value, index)` | `value.charCodeAt(index)` | +| `lib/cpf.ts`, `lib/cnpj.ts` | `str.codeAt(value, i)` | `value.charCodeAt(i)` | +| `lib/cnpj.ts` (`cnpjCheckDigit`) | `seq.get(weights, index)` | `weights[index]` | +| `lib/easter.ts` | `int.min(int.max(…))` | `Math.min(Math.max(…))` | +| `lib/format.ts` (`groupThousands`) | `seq.at(scalars, index) ?? 48` | `scalars[index] ?? 48` | +| `lib/json.ts` | `seq.at(points, i) ?? d` (most call sites) | `points[i] ?? d` | +| `lib/random.ts` | `str.fromInt(randomBelow(10))` | `String(randomBelow(10))` | +| `format-currency.ts` | `str.padStart(str.fromInt(n), …)`, `int.max`, `str.slice` | `String(n).padStart(…)`, `Math.max`, `.slice(…)` | +| `generate-cpf.ts`, `generate-cnpj.ts` | `str.fromInt(n)` | `String(n)` | +| `is-valid-cpf.ts`, `is-valid-cnpj.ts`, `get-address-info-by-cep.ts` | `re.test(PATTERN, s)` | `PATTERN.test(s)` | +| `is-valid-cnpj.ts` | `str.trim(s)` | `s.trim()` | + +`digitAt`, `isRepeated`, `isRepeatedCnpj`, `isRepeatedRun`, `hasValidCnpjChecksum` and every +`.charCodeAt`/`.slice`/`.trim`/`PATTERN.test` call above type-check under real `tsc` +(`core/tsconfig.json`) and produce byte-identical `core/out` (see "Did `core/out` move" below). + +Two ambient declarations were missing outright — `seq.at` and `date.clampEpochDays` — so `tsc` +rejected `lib/json.ts` and `lib/civil.ts` even before this migration touched them (the intrinsics +have always existed; only the hand-maintained `.d.ts` was incomplete). Both are added to +`engine/prelude/index.d.ts`. This is a declarations-only file ("never executed", per its own +header) with no effect on the checker, the generated code or `core/out`; it only makes the +existing "tsc accepts it" claim actually true for code nobody had migrated yet. + +### The checked accessors moved too, in the second pass + +The first pass left four call sites on the namespace form because the bracket idiom's rule — +unchecked when the index is proven in range, checked otherwise — picked the *wrong* accessor for +each of them: the index was provably in range, but the source wanted the checked form anyway, and +a fourth accessor (`str.codeAtOpt`) had no bracket form to pick between in the first place. Both +gaps are closed now (`engine/docs/semantics.md` §7.1, "`??` forces the checked accessor" and +"`s[i]?.charCodeAt(0)`"), and all four sites moved: + +| file | namespace form | ordinary form | +| --- | --- | --- | +| `lib/cnpj.ts`, `hasLetter` | `str.codeAtOpt(value, index) ?? 0` | `value[index]?.charCodeAt(0) ?? 0` | +| `lib/format.ts`, `patternSlots`/`formatWithPattern` | `str.charAtOpt(pattern, index) ?? ""` | `pattern[index] ?? ""` | +| `lib/json.ts`, `matchesAt` | `seq.at(needle, offset) ?? -2` | `needle[offset] ?? -2` | +| `lib/digits.ts`, `keepAlphanumeric` | `str.asciiUpper(value.replace(…))` | `value.replace(…).toUpperCase()` | + +The first three were exactly the "provable index, checked form wanted" case: `pattern` and +`needle` are always fixed-length literals at their call sites (`PATTERN`/`OBFUSCATED_PATTERN`, +`` str.codePoints(`"${key}"`) `` for a literal `key` — `"cep"`, `"uf"`, …), so the frontend's own +provability check would have picked the unchecked accessor. `hasLetter`'s index was never +provable at all, which used to be beside the point: there was no ordinary spelling for the +checked *numeric* accessor for a provable index to pick over either. Now there is. + +Both gaps closed by the same two frontend rules, described in full in `engine/docs/semantics.md` +§7.1: + +- **`??` forces the checked accessor.** Under `noUncheckedIndexedAccess`, real TypeScript already + types a bracket index `T | undefined` regardless of what the checker can prove, so `xs[i] ?? + fallback` is the author's own statement that they want the absent case — not a claim about + provability. `engine/src/core/check.ts`'s `logical` now special-cases `??` over an `index` node: + `this.index(node.left, /* forceChecked */ true)` instead of the provability-driven call. This is + what let `pattern[index] ?? ""` and `needle[offset] ?? -2` replace the namespace form without + moving the Core — the two spellings choose the same accessor, `str.charAtOpt`/`seq.at`, either + way; only the *reason* they choose it changed, from "the frontend proved it" to "the author + wrote `??`". `engine/tests/idioms.spec.ts` asserts both directions: `xs[i] ?? fallback` and + `seq.at(xs, i) ?? fallback` compile to identical Core even where `xs[i]` alone (no `??`) would + have picked `seq.get`. +- **`s[i]?.charCodeAt(0)` is `str.codeAtOpt`.** `s.charCodeAt(i)` alone still has no `??` form — + it answers `NaN` past the end, not `undefined`, so `s.charCodeAt(i) ?? fallback` would compile + but never actually take the fallback branch, a silent behavior change this checker refuses + elsewhere (`.replace`, section 7.1) and refuses here too. But `s[i]` alone already answers + `undefined` past the end, and `?.charCodeAt(0)` on it reads the one scalar's code point only + when present — the same case split `str.codeAtOpt` makes. `engine/src/frontend/lower.ts`'s + `specialCall` recognizes exactly this shape (a plain bracket index, literal `0`) and lowers it + directly to the intrinsic call, the same one the namespace form already produced — so, again, + the Core is unchanged, only reached a different way. Oxc wraps any expression containing `?.` + in a `ChainExpression` node, one level up from where every other optional-chain handling lived, + so a `case "ChainExpression": return this.expr(node.expression);` was needed for this (or any) + `?.` recognition to ever see the node it matches against; nothing depended on that case existing + before because `?.` was always rejected regardless of shape. + +Effect on this migration: `hasLetter` now runs for every input, so `isValidCnpj` no longer throws +on it, and `formatWithPattern` now runs unconditionally, so `formatCnpj` runs for every case +(`conformance/run-source.ts` below has the exact numbers, including why `isValidCnpj` is still not +at 100%, which is unrelated to any of the above). + +### `str.asciiUpper`/`str.asciiLower` gained an ordinary spelling too + +Not a checked accessor, but discovered while migrating what the checked-accessor work unblocked: +`keepAlphanumeric` (`lib/digits.ts`) calls `str.asciiUpper` on the result of `.replace(/[^0-9A-Za-z]/g, +"")` — already the ordinary spelling of `re.retain` on a mixed digit/letter class (first pass), +which types its result `Ascii` (`engine/src/core/check.ts`'s `singleClassOf`: every range's high +end is below `0x80`, so the class is `"ascii"`, not `"digits"`). `str.asciiUpper`'s own doc +already said "a proven-ASCII argument unlocks the host's own case mapping" — the intrinsic was +always meant to gain this idiom, it just hadn't yet. `engine/src/core/check.ts` now maps +`toUpperCase`/`toLowerCase` to `str.asciiUpper`/`str.asciiLower` (`STRING_METHODS`) once the +target is proven ASCII (`requireAsciiCase`, the same gate `charCodeAt`/`charAt`/`slice` already +use, `E_UNICODE_CASE` instead of `E_UTF16_POSITION` when it fails): JavaScript's case methods run +full Unicode case folding, which touches scalars outside ASCII an ASCII-only table leaves alone, +and folds those differently again by target — but restricted to ASCII the two are the identical +function. + +### The one call site that still cannot move + +`is-valid-cnpj.ts`'s alphanumeric (`version: "2"`) path, once a value is confirmed alphanumeric +and the right length, checks the raw, unsanitized shape with `CNPJ_FORMAT.test(str.asciiUpper( +trimmed))`. `trimmed` is `cnpj.trim()` on the function's own `string` parameter — raw external +input, with no ASCII proof — so `.toUpperCase()` is not available to it the way it is to +`keepAlphanumeric`'s already-ASCII result. Proving `trimmed` ASCII first (a regex guard, or +reusing `cleaned`) would restructure the check, not respell it, which is out of scope for a +migration that promises not to move `core/out`. This is the one place `isValidCnpj` still throws: +`conformance/run-source.ts` asserts it precisely (405 of 2,032 cases), the same way the resolved +gaps above used to be asserted. + +Turning on `noUncheckedIndexedAccess` (`core/tsconfig.json`) was considered as part of the same +work — it is what makes `xs[i]` genuinely `T | undefined` to real `tsc`, which is the premise the +`??` rule above rests on — and it was off. Adding it changes nothing except one line: +`cnpjCheckDigit`'s `weights[index]` (`lib/cnpj.ts`), where the loop bound is `weights.length` +itself, provable to the engine's own checker but invisible to `tsc` because `List` erases to a +plain `readonly T[]` (`engine/prelude/index.d.ts`). Every other bracket access in `core/source` +already uses `?? fallback`. But `weights[index]` is on `cnpjCheckDigit`, which every single CNPJ +check calls twice (`hasValidCnpjChecksum`) — reverting it to the namespace form to satisfy `tsc`, +the only Core-preserving fix available (`!` is refused outright elsewhere in this subset for +asserting exactly what it cannot prove, and `as` would be the same claim in different spelling), +was tried and measured: it dropped `isValidCnpj` from 1,627/2,032 matched to 415/2,032, because it +reintroduced an ambient-gap throw into the one helper nearly every case reaches. The flag stays +off: the one blind spot it would catch is not a real bug (the engine's own specialization already +proves `weights[index]` in range independently), and the cost of closing it is far larger than +the gap itself. + +### The five effects, and whether "a real import of a real module" is honest here + +The task frames a real import of a real module as categorically different from an ambient +declaration with nothing behind it — an author reaching for a date or decimal library is ordinary +TypeScript, and the engine recognizing a known module's API is not the same kind of lie. Assessed +honestly, per capability: + +**`date.*` — a date library (e.g. Temporal, `date-fns`, `luxon`).** This is the strongest +candidate. A real `Temporal.PlainDate` (or a well-known library's date-only type) has exactly the +shape `CivilDate` wants: no zone, 1-indexed months, and a constructor that throws or clamps instead +of silently rolling over — the three complaints `E_HOST_DATE` raises about `new Date` do not apply +to it. Recognizing a specific, pinned version of `Temporal` or one library's API and lowering it to +`date.*` is plausible future work, not a change of kind: the semantics genuinely match, unlike +`new Date`, where they do not. The obstacle is scope and churn, not soundness: pinning a dependency +inside what is meant to be dependency-free source, and keeping the recognizer synchronized with +that library's exact surface across versions. + +**`dec.*` — a decimal library (e.g. `decimal.js`, `big.js`).** Also plausible in principle: these +libraries do carry an explicit scale and explicit rounding, which is exactly what `dec.*` requires +and `Number` cannot give. The harder part is that `Decimal`'s scale is a *compile-time* type +parameter the checker uses to keep `add`/`sub` type-safe and to size the generated integer in every +target; a real library's scale lives at the value level, checked at runtime if at all. Recognizing +the *values* a call produces is plausible; recognizing the *compile-time scale discipline* is a +larger design question than swapping a call's spelling, closer to admitting a new kind of +refinement than to adding a row to the idiom table. + +**`random.nextU32` — no.** `Math.random()` is already rejected (`E_MATH_RANDOM`) for the reason +section 4 gives: it is a float in `[0, 1)`, and turning that into an unbiased integer needs a +scaling step that would have to round identically across JavaScript, Python, Go and Rust to stay +unbiased — exactly the kind of target-dependent arithmetic the engine exists to keep out. A real +CSPRNG module (`node:crypto`'s `randomInt`, `crypto.getRandomValues`) is closer, since it can +already produce an unbiased integer in a range directly. But it is a *Node* module, not a portable +one: the whole reason `random.nextU32` exists as a capability, rather than a call to `crypto` +inline, is that Python, Go and Rust each generate their own default environment from their own +standard library (`docs/semantics.md` §4) — recognizing `node:crypto` would tie the ordinary +spelling to one target's runtime, which the capability model specifically avoids. Recognizing it +only for the TypeScript target and requiring the namespace form elsewhere would split the idiom +table by target, which the table's whole design (`tests/idioms.spec.ts` proves *one* Core for both +spellings) does not accommodate today. + +**`http.request` — no, for the reason semantics.md already gives.** `fetch` returns a +`Promise` whose status and body are two separate awaits, and it fails by rejecting rather +than by answering absent. `http.request` answers a plain `Option` with the body +already read. These are different *shapes*, not different spellings of the same shape — recognizing +`fetch(...)` would mean silently collapsing `await fetch(url).then(r => r.json())` into a single +synchronous-looking call, which is exactly the kind of "different operation with the same name" +`.replace()` is refused for (section 6/7.1). A library that already returns `Option`-shaped, +already-read responses could in principle be recognized the way a date or decimal library could; +`fetch` itself cannot be, regardless of how it is imported. + +**`task.race` — no, and for a sharper reason than the others.** `Promise.any` is the closest native +shape, but the engine's semantics are explicitly deterministic under a *virtual* clock (`docs/ +semantics.md` §4.2: "first" is virtual completion time, ties broken by task index) so that a race +is reproducible in conformance testing. `Promise.any` races on the real event loop; recognizing it +would mean recognizing a construct whose result depends on real scheduling, which is the one +property the capability model is built to avoid. No import changes that. + +**`clock.sleep`/`clock.millis` — not asked for, but the same shape as the others.** No ordinary +spelling exists for these either (a raw `setTimeout`/`Promise` pair is real JavaScript but is not a +respelling of a synchronous-looking capability call), and they are used in `get-address-info-by-cep.ts`'s +retry loop. Left on the namespace form, same as `http.request`. + +### `[...s]` has two different types depending on which checker you ask + +One more thing worth recording precisely, because it looked at first like a valid migration and +is not: `docs/semantics.md` §7.1 says `[...s]` is `str.codePoints(s)`, and the *engine's own* +checker treats the two as producing identical Core (`tests/idioms.spec.ts`). But `Ascii`/`Digits` +are ambient aliases for `string` (`engine/prelude/index.d.ts`), so *real* `tsc` infers `[...s]` as +`string[]` — an array of one-character substrings — never as `number[]`. `lib/format.ts`'s +`groupThousands` and `lib/json.ts`'s `jsonStringField` both treat the spread's elements as numeric +code points immediately afterward (arithmetic, comparison against numeric constants), so migrating +either would type-check under the engine's checker and fail under real `tsc` +(`Argument of type 'string' is not assignable to parameter of type 'number'`). Both call sites keep +`str.codePoints`, with a comment explaining why, and both are confirmed clean under +`tsc --noEmit -p core/tsconfig.json`. The idiom is sound only where a spread's elements go on to be +used as strings, not as numbers — this codebase has no such site. + +--- + +## The new step: proving the source runs + +`conformance/run-source.ts` imports the migrated utilities directly — `import { isValidCpf } from +"../source/is-valid-cpf"`, no `tsc`, no `cli.ts build` — and calls them, in Node, against the same +vectors `conformance/cases.ts` already holds, run through `sloppy-imports.mjs` exactly as +`conformance/run.ts` is. Each result is compared against the published npm package (`src/`), the +same ground truth `run.ts` compares the reference interpreter against. It is wired into +`engine/scripts/verify.ts` as a new step, `conformance (source, no engine)`, after `conformance`. + +It runs, and asserts on, exactly: + +- **`isValidCpf`** — every case (1,519) runs for real and matches the published package. The step + fails if even one case falls back to the ambient-gap path: this utility has no known gap left, + so a regression here is real. +- **`formatCnpj`** — every case (160) runs for real and matches, now that `formatWithPattern`'s + checked accessor has an ordinary spelling. Asserted the same way as `isValidCpf`: any case + falling back to the ambient-gap path fails the step, since this utility has no known gap left + either. +- **`isValidCnpj`** — every case (2,032) is attempted; 1,627 run for real and match, 405 hit the + documented `str.asciiUpper(trimmed)` gap ("The one call site that still cannot move" above), + confirmed by asserting the exact `ReferenceError` that line throws, not just "something threw". + The step fails if the matched count reaches zero (the runnable part broke) or if the blocked + count reaches zero (the gap silently vanished or the vectors stopped exercising it — either way, + this file's claim about `isValidCnpj` would be stale and needs updating along with the step). + `hasLetter`'s gap, the one this file used to describe here, is gone: it moved to the ordinary + spelling and is no longer in the picture. +- **Everything else is named and skipped, with a reason, in the same file**: `formatCurrency` + (`dec.*`), `generateCpf`/`generateCnpj` (`random.nextU32`), `getHolidays`/`isBusinessDay` + (`date.*`), and `getAddressInfoByCep` (`http.request`/`task.race`, and it would need a live + network to compare against the published package regardless). + +A mismatch, an unexpected throw, or either count-shrinking condition fails the step and the +overall `verify.ts` run. + +--- + +## Did `core/out` move? + +No generated program logic did, in any of the four targets, in either idiom mode, in either pass. +After this second pass: `node ../engine/src/cli.ts build --project .` (idiomatic) and +`--no-idioms` both regenerate byte-identical `.ts`/`.py`/`.go`/`.rs` files and byte-identical +`API.json`/`LOWERING.md`. + +`SOURCEMAP.json` in each of the four tracked output directories *does* differ — every entry is a +`{start, end}` byte offset into the (now differently spelled, differently commented) source file, +one of the "three review artifacts every target produces" (`backend/generate.ts`) for pointing a +reader from generated code back to its source span. It carries no executable meaning and is +unavoidable: any edit to the source text at all, including a comment-only change, shifts byte +offsets later in the same file. This was checked file by file (`git diff --stat core/out`) rather +than assumed — the four `SOURCEMAP.json` files are the entire diff. + +--- + +## Results + +- `cd core && node ../engine/scripts/verify.ts .` — all 16 steps pass, including + `conformance (source, no engine)`. +- `node engine/scripts/fuzz.ts fast --seed 20260921 --count 1000` — clean (976/1000 compiled, every + produced value stayed inside its proven bounds). +- `node engine/scripts/fuzz.ts full --seed 555 --count 200` — clean (194/200 compiled, interpreter + and every target agree, in both idiom modes). +- `conformance/run.ts` — **4256/4256** matched, per target, in both idiom modes (typescript, + python, go, rust, and their `-plain` counterparts), unchanged from before this pass. +- `conformance/run-source.ts` — `isValidCpf` 1,519/1,519 (unchanged), `formatCnpj` 160/160 (newly + covered — it did not run at all before this pass), `isValidCnpj` 1,627/2,032 (405 on the one + remaining `str.asciiUpper(trimmed)` gap, up from 1,216/2,032 before). Six utilities still do not + run — `formatCurrency`, `generateCpf`, `generateCnpj`, `getHolidays`, `isBusinessDay`, + `getAddressInfoByCep` — none of them for a reason this pass touches. diff --git a/core/docs/survey.md b/core/docs/survey.md new file mode 100644 index 000000000..33430d4d7 --- /dev/null +++ b/core/docs/survey.md @@ -0,0 +1,194 @@ +# Survey of the published package + +Generated by `node scripts/survey.ts`. The classification is syntactic and conservative: it +reads each utility and reports the features it appears to need. It answers the question the +engine exists to answer — how many utilities could be authored once — with a number rather +than a feeling. + +**138 utilities.** 120 use only features the engine supports today; +18 touch something that is not admitted yet. + +## Features by utility count + +| feature | utilities | status | +| --- | --- | --- | +| strings | 138 | supported | +| dataset | 105 | supported as constant tables; a baked dataset type is not admitted yet | +| regex | 35 | supported (explicit classes only) | +| float | 34 | supported | +| ascii | 33 | supported | +| random | 13 | supported (PCG32 in source over random.nextU32) | +| maps | 10 | blocked: Map and Set are not admitted yet | +| dates | 9 | supported (CivilDate) | +| unions | 5 | blocked: discriminated unions are not admitted yet | +| collation | 4 | supported (scalar order); locale collation is not admitted | +| unicode | 4 | blocked: normalization and locale case mapping are out of scope | +| decimal | 3 | supported (fixed scale) | +| http | 2 | supported | + +## Blocked utilities + +| utility | blocked by | +| --- | --- | +| `capitalize` | maps | +| `format-phone` | maps | +| `generate-boleto` | unions | +| `generate-legal-nature` | maps | +| `get-area-codes-by-state` | maps | +| `get-boleto-info` | unions | +| `get-certidao-info` | unions | +| `get-cities` | unicode | +| `get-holidays` | maps, unions | +| `get-legal-natures` | maps | +| `get-legal-natures-by-category` | maps | +| `get-municipalities` | unicode | +| `get-municipality` | maps | +| `get-pix-key-info` | unions | +| `get-states` | unicode | +| `is-business-day` | maps | +| `is-valid-ncm` | maps | +| `remove-accents` | unicode | + +## Every utility + +| utility | lines | features | +| --- | --- | --- | +| `add-business-days` | 30 | dates, float, strings | +| `capitalize` | 133 | dataset, maps, regex, strings | +| `convert-currency-to-words` | 38 | float, strings | +| `convert-date-to-words` | 66 | dataset, dates, float, regex, strings | +| `convert-license-plate-to-mercosul` | 10 | ascii, dataset, strings | +| `convert-number-to-words` | 21 | float, strings | +| `difference-in-business-days` | 34 | dates, strings | +| `format-boleto` | 22 | dataset, strings | +| `format-caepf` | 18 | dataset, strings | +| `format-cei` | 18 | dataset, strings | +| `format-cep` | 17 | strings | +| `format-certidao` | 18 | dataset, strings | +| `format-cnae` | 17 | dataset, strings | +| `format-cnh` | 17 | strings | +| `format-cno` | 18 | dataset, strings | +| `format-cnpj` | 22 | dataset, strings | +| `format-cns` | 17 | strings | +| `format-cpf` | 20 | dataset, strings | +| `format-currency` | 45 | decimal, float, strings | +| `format-iban` | 14 | dataset, strings | +| `format-legal-nature` | 20 | strings | +| `format-license-plate` | 19 | dataset, regex, strings | +| `format-ncm` | 17 | strings | +| `format-nfe-key` | 14 | dataset, strings | +| `format-passport` | 3 | strings | +| `format-phone` | 76 | dataset, maps, strings | +| `format-pis` | 17 | strings | +| `format-processo-juridico` | 20 | strings | +| `format-voter-id` | 15 | dataset, strings | +| `generate-boleto` | 59 | dataset, float, random, strings, unions | +| `generate-cep` | 3 | float, random, strings | +| `generate-cnh` | 14 | float, random, strings | +| `generate-cnpj` | 67 | ascii, dataset, float, random, strings | +| `generate-cpf` | 21 | dataset, float, random, strings | +| `generate-legal-nature` | 7 | dataset, float, maps, random, strings | +| `generate-license-plate` | 20 | ascii, float, random, strings | +| `generate-passport` | 10 | dataset, float, random, strings | +| `generate-phone` | 37 | dataset, float, random, strings | +| `generate-pis` | 11 | float, random, strings | +| `generate-pix-payload` | 157 | dataset, decimal, float, regex, strings | +| `generate-processo-juridico` | 35 | ascii, dataset, dates, float, random, strings | +| `generate-renavam` | 12 | float, random, strings | +| `generate-voter-id` | 22 | dataset, float, random, strings | +| `get-address-info-by-cep` | 184 | ascii, http, regex, strings | +| `get-area-code-info` | 38 | dataset, float, strings | +| `get-area-codes-by-state` | 15 | ascii, dataset, float, maps, strings | +| `get-bank-by-code` | 15 | ascii, dataset, strings | +| `get-bank-by-ispb` | 15 | ascii, dataset, strings | +| `get-banks` | 4 | dataset, strings | +| `get-boleto-info` | 92 | dataset, dates, float, strings, unions | +| `get-cbo` | 22 | dataset, regex, strings | +| `get-cep-info-by-address` | 99 | ascii, http, strings | +| `get-certidao-info` | 57 | ascii, dataset, float, strings, unions | +| `get-cfop` | 20 | dataset, regex, strings | +| `get-cities` | 16 | collation, dataset, strings, unicode | +| `get-cnae` | 22 | dataset, regex, strings | +| `get-format-license-plate` | 12 | dataset, regex, strings | +| `get-holidays` | 136 | collation, dataset, dates, maps, strings, unions | +| `get-iban-info` | 45 | ascii, strings | +| `get-legal-nature` | 45 | dataset, strings | +| `get-legal-natures` | 16 | dataset, maps, strings | +| `get-legal-natures-by-category` | 26 | dataset, maps, strings | +| `get-municipalities` | 16 | collation, dataset, strings, unicode | +| `get-municipality` | 84 | ascii, dataset, maps, strings | +| `get-municipality-by-code` | 19 | dataset, strings | +| `get-nfe-key-info` | 96 | dataset, float, regex, strings | +| `get-pix-key-info` | 46 | ascii, dataset, regex, strings, unions | +| `get-pix-payload-info` | 186 | ascii, dataset, float, regex, strings | +| `get-state-by-ibge-code` | 12 | dataset, float, strings | +| `get-state-code-by-name` | 11 | ascii, dataset, regex, strings | +| `get-state-name-by-code` | 9 | ascii, dataset, strings | +| `get-states` | 4 | collation, dataset, strings, unicode | +| `get-timezone-by-state` | 7 | ascii, dataset, strings | +| `is-business-day` | 29 | dataset, dates, maps, strings | +| `is-holiday` | 31 | dataset, dates, strings | +| `is-valid-bank-account` | 223 | ascii, dataset, regex, strings | +| `is-valid-boleto` | 36 | ascii, dataset, strings | +| `is-valid-caepf` | 24 | dataset, float, regex, strings | +| `is-valid-cbo` | 3 | strings | +| `is-valid-cei` | 3 | float, strings | +| `is-valid-cep` | 7 | dataset, regex, strings | +| `is-valid-certidao` | 41 | ascii, dataset, regex, strings | +| `is-valid-cfop` | 10 | dataset, regex, strings | +| `is-valid-cnae` | 3 | dataset, strings | +| `is-valid-cnh` | 17 | ascii, dataset, regex, strings | +| `is-valid-cno` | 3 | float, strings | +| `is-valid-cnpj` | 32 | ascii, dataset, regex, strings | +| `is-valid-cns` | 32 | dataset, regex, strings | +| `is-valid-cpf` | 15 | ascii, regex, strings | +| `is-valid-credit-card` | 16 | ascii, dataset, regex, strings | +| `is-valid-csosn` | 10 | dataset, regex, strings | +| `is-valid-cst` | 39 | ascii, dataset, regex, strings | +| `is-valid-email` | 7 | regex, strings | +| `is-valid-iban` | 22 | ascii, dataset, regex, strings | +| `is-valid-ie` | 375 | ascii, dataset, regex, strings | +| `is-valid-landline-phone` | 16 | ascii, dataset, float, strings | +| `is-valid-legal-nature` | 8 | dataset, strings | +| `is-valid-license-plate` | 4 | strings | +| `is-valid-mobile-phone` | 31 | ascii, dataset, float, strings | +| `is-valid-ncm` | 17 | dataset, maps, regex, strings | +| `is-valid-nfe-key` | 3 | strings | +| `is-valid-passport` | 7 | dataset, regex, strings | +| `is-valid-phone` | 37 | dataset, strings | +| `is-valid-pis` | 14 | ascii, dataset, regex, strings | +| `is-valid-pix-key` | 13 | strings | +| `is-valid-pix-payload` | 3 | strings | +| `is-valid-processo-juridico` | 36 | ascii, dataset, float, regex, strings | +| `is-valid-registro-profissional` | 50 | dataset, float, regex, strings | +| `is-valid-renavam` | 17 | ascii, dataset, regex, strings | +| `is-valid-service-phone` | 28 | dataset, strings | +| `is-valid-vin` | 25 | ascii, dataset, strings | +| `is-valid-voter-id` | 23 | dataset, float, regex, strings | +| `parse-boleto` | 11 | dataset, strings | +| `parse-caepf` | 6 | dataset, strings | +| `parse-cbo` | 6 | dataset, strings | +| `parse-cei` | 6 | dataset, strings | +| `parse-cep` | 6 | dataset, strings | +| `parse-certidao` | 6 | dataset, strings | +| `parse-cfop` | 6 | dataset, strings | +| `parse-cnae` | 6 | dataset, strings | +| `parse-cnh` | 6 | dataset, strings | +| `parse-cno` | 6 | dataset, strings | +| `parse-cnpj` | 9 | dataset, strings | +| `parse-cns` | 6 | dataset, strings | +| `parse-cpf` | 6 | dataset, strings | +| `parse-currency` | 15 | decimal, strings | +| `parse-iban` | 6 | dataset, strings | +| `parse-legal-nature` | 6 | dataset, strings | +| `parse-license-plate` | 7 | dataset, strings | +| `parse-ncm` | 6 | dataset, strings | +| `parse-nfe-key` | 9 | dataset, strings | +| `parse-passport` | 5 | dataset, strings | +| `parse-phone` | 6 | dataset, strings | +| `parse-pis` | 6 | dataset, strings | +| `parse-processo-juridico` | 6 | dataset, strings | +| `parse-voter-id` | 12 | dataset, strings | +| `remove-accents` | 6 | strings, unicode | +| `sub-business-days` | 12 | dates, strings | + diff --git a/core/engine.config.json b/core/engine.config.json new file mode 100644 index 000000000..c931d9d6d --- /dev/null +++ b/core/engine.config.json @@ -0,0 +1,12 @@ +{ + "name": "brazilian-utils-core", + "sourceRoot": "source", + "out": "out", + "targets": ["typescript", "python", "go", "rust"], + "packages": { + "typescript": "core", + "python": "_core", + "go": "internal/core", + "rust": "coreout" + } +} diff --git a/core/out/go/API.json b/core/out/go/API.json new file mode 100644 index 000000000..a22441db0 --- /dev/null +++ b/core/out/go/API.json @@ -0,0 +1,252 @@ +{ + "functions": [ + { + "name": "FormatCnpj", + "module": "format-cnpj.go", + "params": [ + { + "name": "value", + "type": "string" + }, + { + "name": "options", + "type": "FormatCnpjOptions" + } + ], + "returns": "string", + "effects": [] + }, + { + "name": "FormatCurrency", + "module": "format-currency.go", + "params": [ + { + "name": "value", + "type": "int" + }, + { + "name": "symbol", + "type": "bool" + } + ], + "returns": "string", + "effects": [] + }, + { + "name": "GenerateCnpj", + "module": "generate-cnpj.go", + "params": [ + { + "name": "env", + "type": "Capabilities" + } + ], + "returns": "string", + "effects": [ + "env" + ] + }, + { + "name": "GenerateCpf", + "module": "generate-cpf.go", + "params": [ + { + "name": "env", + "type": "Capabilities" + } + ], + "returns": "string", + "effects": [ + "env" + ] + }, + { + "name": "GetAddressInfoByCep", + "module": "get-address-info-by-cep.go", + "params": [ + { + "name": "cep", + "type": "string" + }, + { + "name": "env", + "type": "Capabilities" + } + ], + "returns": "AddressInfo", + "effects": [ + "Fail", + "Fail", + "env" + ] + }, + { + "name": "GetHolidays", + "module": "get-holidays.go", + "params": [ + { + "name": "year", + "type": "int" + } + ], + "returns": "[]Holiday", + "effects": [] + }, + { + "name": "IsBusinessDay", + "module": "is-business-day.go", + "params": [ + { + "name": "value", + "type": "int" + }, + { + "name": "includeOptional", + "type": "bool" + } + ], + "returns": "bool", + "effects": [] + }, + { + "name": "IsValidCnpj", + "module": "is-valid-cnpj.go", + "params": [ + { + "name": "cnpj", + "type": "string" + }, + { + "name": "version", + "type": "string" + } + ], + "returns": "bool", + "effects": [] + }, + { + "name": "IsValidCpf", + "module": "is-valid-cpf.go", + "params": [ + { + "name": "cpf", + "type": "string" + } + ], + "returns": "bool", + "effects": [] + } + ], + "seams": [ + { + "name": "GenerateCnpj", + "publicName": "GenerateCnpj", + "module": "generate-cnpj.go", + "params": [ + { + "name": "env", + "type": "Capabilities" + } + ], + "returns": "string", + "hasWrapper": false + }, + { + "name": "GenerateCpf", + "publicName": "GenerateCpf", + "module": "generate-cpf.go", + "params": [ + { + "name": "env", + "type": "Capabilities" + } + ], + "returns": "string", + "hasWrapper": false + }, + { + "name": "GetAddressInfoByCep", + "publicName": "GetAddressInfoByCep", + "module": "get-address-info-by-cep.go", + "params": [ + { + "name": "cep", + "type": "string" + }, + { + "name": "env", + "type": "Capabilities" + } + ], + "returns": "AddressInfo", + "hasWrapper": false + } + ], + "records": [ + { + "name": "FormatCnpjOptions", + "fields": [ + { + "name": "Pad", + "type": "bool" + }, + { + "name": "Version", + "type": "string" + }, + { + "name": "Obfuscate", + "type": "bool" + } + ] + }, + { + "name": "AddressInfo", + "fields": [ + { + "name": "Cep", + "type": "string" + }, + { + "name": "State", + "type": "string" + }, + { + "name": "City", + "type": "string" + }, + { + "name": "Neighborhood", + "type": "string" + }, + { + "name": "Street", + "type": "string" + } + ] + }, + { + "name": "Holiday", + "fields": [ + { + "name": "Name", + "type": "string" + }, + { + "name": "Date", + "type": "int" + }, + { + "name": "Type", + "type": "string" + } + ] + } + ], + "errors": [ + "GetAddressInfoByCepError", + "GetAddressInfoByCepNotFoundError", + "GetAddressInfoByCepValidationError", + "HttpError" + ] +} diff --git a/core/out/go/LOWERING.md b/core/out/go/LOWERING.md new file mode 100644 index 000000000..4b4728392 --- /dev/null +++ b/core/out/go/LOWERING.md @@ -0,0 +1,323 @@ +# Lowering selections — go + +Generated by the engine. Each row is one operation, the argument types it was called +with, the implementation that was selected, and the rule that decided it. + +| operation | argument types | implementation | why | +| --- | --- | --- | --- | +| `clock.sleep` | `Duration` | native | only candidate, cost none/constant | +| `core.eq` | `"1" \| "2", "2"` | native | only candidate, cost none/constant | +| `core.eq` | `"national" \| "optional" \| "religious" \| "state", "optional"` | native | only candidate, cost none/constant | +| `core.eq` | `Ascii[1], Ascii[1]` | native | only candidate, cost none/constant | +| `core.eq` | `Ascii[1], Digits[1]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[-1..1], Int[0..0]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[-2..2], Int[0..0]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[-48..79], Int[0..9]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[-1..-1]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[0..127]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[10..10]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[110..110]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[114..114]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[116..116]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[117..117]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[13..13]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[32..32]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[34..34]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[58..58]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[9..9]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[92..92]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..2147483647], Int[11..11]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..2147483647], Int[14..14]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..3506328], Int[306..-1]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..9], Int[0..9]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..9600], Int[0..-1]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..12], Int[1..12]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..31], Int[1..31]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..7], Int[6..6]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..7], Int[7..7]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..9999], Int[1..9999]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[48..57], Int[48..57]` | native | only candidate, cost none/constant | +| `core.eq` | `String[0..2147483647], Ascii[0]` | native | only candidate, cost none/constant | +| `date.addDays` | `CivilDate, Int[-2..-2]` | library | only candidate, cost none/constant | +| `date.addDays` | `CivilDate, Int[-47..-47]` | library | only candidate, cost none/constant | +| `date.addDays` | `CivilDate, Int[60..60]` | library | only candidate, cost none/constant | +| `date.clampEpochDays` | `Int[0..0]` | native | only candidate, cost none/constant | +| `date.compare` | `CivilDate, CivilDate` | library | only candidate, cost none/constant | +| `date.dayOfWeek` | `CivilDate` | native | only candidate, cost none/constant | +| `date.fromYmd` | `Int[1900..2099], Int[1..1], Int[1..1]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[10..10], Int[12..12]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[11..11], Int[15..15]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[11..11], Int[2..2]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[12..12], Int[25..25]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[3..3], Int[22..31]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[4..4], Int[1..25]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[4..4], Int[21..21]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[5..5], Int[1..1]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[9..9], Int[7..7]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[2024..2099], Int[11..11], Int[20..20]` | portable | only candidate, cost none/linear | +| `date.year` | `CivilDate` | portable | only candidate, cost none/constant | +| `dec.abs` | `Decimal<2>` | library | only candidate, cost none/constant | +| `dec.isNegative` | `Decimal<2>` | native | only candidate, cost none/constant | +| `dec.unscaled` | `Decimal<2>` | native | only candidate, cost none/constant | +| `http.request` | `HttpRequest` | native | only candidate, cost many/linear | +| `int.add` | `Int[-10012..20013], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-146097..3506328], Int[-3506503..3798697]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-14618796..14618800], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-238871..9], Int[3..3]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-3504000..3795635], Int[-2400..2599]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-3506503..3798330], Int[0..367]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-3508380..3800745], Int[-2403..2603]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-3508623..3800862], Int[-95..103]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-36547263..36546651], Int[2..2]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-36547330..36546740], Int[2..2]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-7..35], Int[114..114]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-719162..2932896], Int[719468..719468]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-9612..10413], Int[-400..9600]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..1683], Int[2..2]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..17], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..18], Int[0..319]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..2147483646], Int[0..4]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..2147483646], Int[5..5]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..29], Int[0..6]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..30], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..337], Int[0..132]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..337], Int[1..31]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..342], Int[19..20]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..65520], Int[0..15]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..720], Int[0..90]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..891], Int[0..81]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..891], Int[0..99]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..9007199254740991], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..9007199254740991], Int[2..2]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..9007199254740991], Int[6..6]` | native | only candidate, cost none/constant | +| `int.add` | `Int[1..2], Int[9..9]` | native | only candidate, cost none/constant | +| `int.add` | `Int[1..31], Int[0..31]` | native | only candidate, cost none/constant | +| `int.add` | `Int[32..32], Int[0..6]` | native | only candidate, cost none/constant | +| `int.add` | `Int[32..38], Int[0..48]` | native | only candidate, cost none/constant | +| `int.add` | `Int[5..2147483658], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[5..2147483659], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[6..2147483667], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[6..2147483668], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[8..352], Int[15..15]` | native | only candidate, cost none/constant | +| `int.add` | `Int[9..2147483671], Int[0..3]` | native | only candidate, cost none/constant | +| `int.div` | `Int[-3506022..3798461], Int[1460..1460]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[-3506022..3798461], Int[146096..146096]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[-3506022..3798461], Int[36524..36524]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[-3508743..3800988], Int[365..365]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[-36547261..36546653], Int[5..5]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[-36547328..36546742], Int[153..153]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[-9600..10399], Int[100..100]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[-9600..10399], Int[4..4]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[-9612..10413], Int[100..100]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[-9612..10413], Int[4..4]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[0..469], Int[451..451]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[0..99], Int[4..4]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[0..9999], Int[400..400]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[107..149], Int[31..31]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[19..20], Int[4..4]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[1900..2099], Int[100..100]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[2..1685], Int[5..5]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.div` | `Int[306..3652364], Int[146097..146097]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.ge` | `Int[0..1114111], Int[0..0]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..1114111], Int[48..48]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..1114111], Int[65..65]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..1114111], Int[97..97]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..127], Int[65..65]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..17], Int[0..2147483647]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..599], Int[200..200]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[1900..2099], Int[2024..2024]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[0..14], Int[0..0]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[0..2], Int[0..0]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1..12], Int[2..2]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1..9007199254740991], Int[12..12]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1..9007199254740991], Int[31..31]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1..9007199254740991], Int[9999..9999]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1900..9999], Int[2099..2099]` | native | only candidate, cost none/constant | +| `int.le` | `Int[-238868..238858], Int[2..2]` | native | only candidate, cost none/constant | +| `int.le` | `Int[1..12], Int[2..2]` | native | only candidate, cost none/constant | +| `int.le` | `Int[22..56], Int[31..31]` | native | only candidate, cost none/constant | +| `int.le` | `Int[48..1114111], Int[57..57]` | native | only candidate, cost none/constant | +| `int.le` | `Int[65..1114111], Int[70..70]` | native | only candidate, cost none/constant | +| `int.le` | `Int[65..127], Int[90..90]` | native | only candidate, cost none/constant | +| `int.le` | `Int[97..1114111], Int[102..102]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[-238871..238867], Int[10..10]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[-9007199254740991..9007199254740991], Int[1..1]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[0..10], Int[2..2]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[0..17], Int[0..2147483647]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[0..4294967295], Int[4294967287..4294967296]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[0..9999], Int[0..0]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[1..9999], Int[1900..1900]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[200..599], Int[300..300]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[306..3652364], Int[0..0]` | native | only candidate, cost none/constant | +| `int.max` | `Int[-10012..20014], Int[1..1]` | native | only candidate, cost none/constant | +| `int.max` | `Int[-14618795..14618801], Int[1..1]` | native | only candidate, cost none/constant | +| `int.max` | `Int[-238868..238858], Int[1..1]` | native | only candidate, cost none/constant | +| `int.max` | `Int[-4372068..6585557], Int[-719162..-719162]` | native | only candidate, cost none/constant | +| `int.max` | `Int[1..15], Int[0..0]` | native | only candidate, cost none/constant | +| `int.max` | `Int[1..62], Int[22..22]` | native | only candidate, cost none/constant | +| `int.min` | `Int[-719162..6585557], Int[2932896..2932896]` | native | only candidate, cost none/constant | +| `int.min` | `Int[0..65535], Int[65535..65535]` | native | only candidate, cost none/constant | +| `int.min` | `Int[1..14618801], Int[31..31]` | native | only candidate, cost none/constant | +| `int.min` | `Int[1..20014], Int[9999..9999]` | native | only candidate, cost none/constant | +| `int.min` | `Int[1..238858], Int[12..12]` | native | only candidate, cost none/constant | +| `int.min` | `Int[22..62], Int[56..56]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[-14..14], Int[3..3]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.mod` | `Int[0..4294967295], Int[10..10]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.mod` | `Int[0..810], Int[11..11]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.mod` | `Int[0..86], Int[7..7]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.mod` | `Int[0..972], Int[11..11]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.mod` | `Int[0..99], Int[4..4]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.mod` | `Int[0..990], Int[11..11]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.mod` | `Int[107..149], Int[31..31]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.mod` | `Int[19..20], Int[4..4]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.mod` | `Int[1900..2099], Int[100..100]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.mod` | `Int[1900..2099], Int[19..19]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.mod` | `Int[23..367], Int[30..30]` | native | only candidate, cost none/constant; Go truncates, which is the Core's rule | +| `int.mul` | `Int[-1..24], Int[146097..146097]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[-1..24], Int[400..400]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[-9600..10399], Int[365..365]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..1], Int[31..31]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..24], Int[146097..146097]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..24], Int[400..400]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..4095], Int[16..16]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..9], Int[2..10]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..9], Int[2..11]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..9], Int[2..9]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[11..11], Int[0..29]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[153..153], Int[-238871..238867]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[153..153], Int[0..11]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[19..19], Int[0..18]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[2..2], Int[0..24]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[2..2], Int[0..3]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[22..22], Int[0..6]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[365..365], Int[-9612..10413]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[5..5], Int[-7309466..7309348]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[7..7], Int[0..1]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3506022..3798461], Int[-2401..2601]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3506022..3798461], Int[-3510887..3803444]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3506400..3798234], Int[-96..103]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3508718..3800965], Int[-23..25]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3510783..3803348], Int[-96..104]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3652600..7305025], Int[719468..719468]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-7309466..7309348], Int[-7309452..7309330]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..127], Int[48..48]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..15], Int[1..14]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..24], Int[1..1]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..35], Int[0..7]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..9999], Int[-400..9600]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[1..368], Int[1..1]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[1..9999], Int[1..1]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[10..10], Int[0..8]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[10..238867], Int[9..9]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[11..11], Int[0..9]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[11..11], Int[2..10]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[14..358], Int[6..6]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[19..362], Int[4..5]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[3..12], Int[3..3]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[3..17], Int[2..2]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[3..4], Int[3..3]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[3..86], Int[0..3]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[306..3652364], Int[-146097..3506328]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[32..56], Int[31..31]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[32..86], Int[0..29]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[48..57], Int[48..48]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[65..70], Int[55..55]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[97..102], Int[87..87]` | native | only candidate, cost none/constant | +| `opt.isNone` | `Option` | native | only candidate, cost none/constant | +| `opt.isNone` | `Option` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Ascii[0]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, CivilDate` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[-1..-1]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[0..0]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[48..48]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[-2..-2]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[0..0]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[48..48]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Ascii[0]` | native | only candidate, cost none/constant | +| `opt.unwrap` | `Option` | native | only candidate, cost none/constant | +| `opt.unwrap` | `Option` | native | only candidate, cost none/constant | +| `random.nextU32` | `` | native | only candidate, cost none/constant | +| `re.retain` | `String[0..2147483647]` | native | only candidate, cost one/linear; a byte-wise scan when every retained range is ASCII | +| `re.test` | `String[0..2147483647]` | native | only candidate, cost none/linear; the normalized pattern is inside the compatibility subset; \A…\z anchors the whole string | +| `seq.at` | `List[0..2147483647], Int[0..2147483650]` | library | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[0..9007199254740991]` | library | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[1..9007199254740992]` | library | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[5..2147483658]` | library | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[5..2147483659]` | library | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[6..2147483667]` | library | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[6..2147483668]` | library | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[9..2147483674]` | library | only candidate, cost none/constant | +| `seq.at` | `List[5..5], Int[0..4]` | library | only candidate, cost none/constant | +| `seq.at` | `List[0..15], Int[0..14]` | library | only candidate, cost none/constant | +| `seq.get` | `List[12..12], Int[0..11]` | native | only candidate, cost none/constant | +| `seq.len` | `List[0..2147483647]` | native | only candidate, cost none/constant | +| `seq.len` | `List[5..5]` | native | only candidate, cost none/constant | +| `seq.len` | `List[12..12]` | native | only candidate, cost none/constant | +| `seq.len` | `List[0..15]` | native | only candidate, cost none/constant | +| `seq.push` | `` | native | only candidate, cost none/constant | +| `seq.sortStableBy` | `List[12..13], (Holiday) => CivilDate` | native | only candidate, cost one/nlogn | +| `str.asciiUpper` | `Ascii[0..2147483647]` | native | only candidate, cost one/linear; strings.ToUpper is only ASCII-equivalent on ASCII input | +| `str.asciiUpper` | `String[0..2147483647]` | native | native, cost one/linear; strings.Map is one pass, and mapping only a-z is the Core's rule for any input; rejected strings.ToUpper is only ASCII-equivalent on ASCII input | +| `str.charAtOpt` | `Ascii[0..2147483647], Int[0..17]` | library | only candidate, cost one/linear | +| `str.charAtOpt` | `Ascii[18], Int[0..17]` | library | only candidate, cost one/linear | +| `str.codeAt` | `Ascii[14], Int[12..12]` | native | only candidate, cost none/constant; indexing a string yields a byte | +| `str.codeAt` | `Ascii[14], Int[13..13]` | native | only candidate, cost none/constant; indexing a string yields a byte | +| `str.codeAt` | `Digits[11], Int[0..0]` | native | only candidate, cost none/constant; indexing a string yields a byte | +| `str.codeAt` | `Digits[11], Int[0..8]` | native | only candidate, cost none/constant; indexing a string yields a byte | +| `str.codeAt` | `Digits[11], Int[1..10]` | native | only candidate, cost none/constant; indexing a string yields a byte | +| `str.codeAt` | `Digits[11], Int[10..10]` | native | only candidate, cost none/constant; indexing a string yields a byte | +| `str.codeAt` | `Digits[11], Int[9..9]` | native | only candidate, cost none/constant; indexing a string yields a byte | +| `str.codeAt` | `Digits[12], Int[0..0]` | native | only candidate, cost none/constant; indexing a string yields a byte | +| `str.codeAt` | `Digits[12], Int[1..11]` | native | only candidate, cost none/constant; indexing a string yields a byte | +| `str.codeAt` | `Digits[14], Int[0..0]` | native | only candidate, cost none/constant; indexing a string yields a byte | +| `str.codeAt` | `Digits[14], Int[0..11]` | native | only candidate, cost none/constant; indexing a string yields a byte | +| `str.codeAt` | `Digits[14], Int[1..13]` | native | only candidate, cost none/constant; indexing a string yields a byte | +| `str.codeAtOpt` | `Ascii[0..2147483647], Int[0..2147483646]` | library | only candidate, cost none/constant | +| `str.codePoints` | `Ascii[5]` | native | only candidate, cost one/linear; an ASCII byte is already its own code point, so `[]byte(value)` needs no UTF-8 decode at all, unlike `codePoints`' `range` over the string below | +| `str.codePoints` | `Digits[0..15]` | native | only candidate, cost one/linear; an ASCII byte is already its own code point, so `[]byte(value)` needs no UTF-8 decode at all, unlike `codePoints`' `range` over the string below | +| `str.codePoints` | `String[0..2147483647]` | library | library, cost one/linear; rejected an ASCII byte is already its own code point, so `[]byte(value)` needs no UTF-8 decode at all, unlike `codePoints`' `range` over the string below | +| `str.concat` | `Ascii[0..17], Ascii[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[0..3], Ascii[1..47]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[0..30], Ascii[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[1..31], Ascii[0..16]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[1..4], Ascii[1..47]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[1], Ascii[0..3]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[1], Ascii[3]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[2], Ascii[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[25], Digits[8] matches ^[0-9]{8}$` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[33], Ascii[6]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[36], Digits[8] matches ^[0-9]{8}$` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[4], Ascii[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[1], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[10], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[11], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[12], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[12], Digits[2]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[13], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[2], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[3], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[4], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[5], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[6], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[7], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[8], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[9], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[9], Digits[2]` | native | only candidate, cost one/linear | +| `str.fromCodePoints` | `List[0..2147483647]` | library | library, cost one/linear; rejected every code point this project ever builds this way is proven ASCII (`group_thousands`' `out`, `engine/docs/progress.md` §8), so its byte value is its whole UTF-8 encoding -- one []byte built directly and converted once, instead of `fromCodePoints`' []rune round trip below, which lets Go's own UTF-8 encoder re-derive what a byte already was | +| `str.fromCodePoints` | `List[0..30]` | native | only candidate, cost one/linear; every code point this project ever builds this way is proven ASCII (`group_thousands`' `out`, `engine/docs/progress.md` §8), so its byte value is its whole UTF-8 encoding -- one []byte built directly and converted once, instead of `fromCodePoints`' []rune round trip below, which lets Go's own UTF-8 encoder re-derive what a byte already was | +| `str.fromCodePoints` | `List[1..1]` | native | only candidate, cost one/linear; every code point this project ever builds this way is proven ASCII (`group_thousands`' `out`, `engine/docs/progress.md` §8), so its byte value is its whole UTF-8 encoding -- one []byte built directly and converted once, instead of `fromCodePoints`' []rune round trip below, which lets Go's own UTF-8 encoder re-derive what a byte already was | +| `str.fromInt` | `Int[-9007199254740991..9007199254740991]` | native | only candidate, cost one/linear | +| `str.fromInt` | `Int[0..9]` | native | only candidate, cost one/linear | +| `str.len` | `Ascii[0..2147483647]` | native | only candidate, cost none/constant; `len` counts bytes, which equals the scalar count only for ASCII | +| `str.len` | `Ascii[18]` | native | only candidate, cost none/constant; `len` counts bytes, which equals the scalar count only for ASCII | +| `str.len` | `Ascii[3..17]` | native | only candidate, cost none/constant; `len` counts bytes, which equals the scalar count only for ASCII | +| `str.len` | `Digits[0..2147483647]` | native | only candidate, cost none/constant; `len` counts bytes, which equals the scalar count only for ASCII | +| `str.len` | `Digits[12]` | native | only candidate, cost none/constant; `len` counts bytes, which equals the scalar count only for ASCII | +| `str.padStart` | `Ascii[0..2147483647], Int[0..18], Digits[1]` | library | only candidate, cost one/linear | +| `str.padStart` | `Ascii[1..17], Int[3..3], Digits[1]` | library | only candidate, cost one/linear | +| `str.slice` | `Ascii[3..17], Int[0..0], Int[1..15]` | native | only candidate, cost none/constant; slicing cuts at byte boundaries | +| `str.slice` | `Ascii[3..17], Int[1..15], Int[3..17]` | native | only candidate, cost none/constant; slicing cuts at byte boundaries | +| `str.trim` | `String[0..2147483647]` | native | only candidate, cost none/constant; strings.Trim takes the cut set explicitly, so the 25 code points are exact | +| `task.race` | `List<() => Option>[2..2]` | library | only candidate, cost many/linear; goroutines with a context and a channel are the idiomatic form | + +Mix: 279 native, 23 library, 12 portable. diff --git a/core/out/go/SOURCEMAP.json b/core/out/go/SOURCEMAP.json new file mode 100644 index 000000000..caaa62e8b --- /dev/null +++ b/core/out/go/SOURCEMAP.json @@ -0,0 +1,287 @@ +{ + "lib_digits.go#keepAlphanumeric": { + "module": "lib/digits", + "start": 575, + "end": 684 + }, + "lib_digits.go#keepDigits": { + "module": "lib/digits", + "start": 397, + "end": 481 + }, + "lib_digits.go#isRepeatedRun": { + "module": "lib/digits", + "start": 1135, + "end": 1358 + }, + "lib_digits.go#digitAt": { + "module": "lib/digits", + "start": 738, + "end": 842 + }, + "lib_digits.go#digitAt2": { + "module": "lib/digits", + "start": 738, + "end": 842 + }, + "lib_digits.go#digitAt3": { + "module": "lib/digits", + "start": 738, + "end": 842 + }, + "lib_format.go#patternSlots": { + "module": "lib/format", + "start": 1345, + "end": 2086 + }, + "lib_format.go#formatWithPattern": { + "module": "lib/format", + "start": 2182, + "end": 2821 + }, + "lib_format.go#groupThousands": { + "module": "lib/format", + "start": 664, + "end": 1279 + }, + "format-cnpj.go#FormatCnpj": { + "module": "format-cnpj", + "start": 967, + "end": 1233 + }, + "format-currency.go#FormatCurrency": { + "module": "format-currency", + "start": 966, + "end": 1499 + }, + "lib_random.go#randomBelow": { + "module": "lib/random", + "start": 1137, + "end": 1640 + }, + "lib_random.go#randomDigit": { + "module": "lib/random", + "start": 1680, + "end": 1747 + }, + "lib_cnpj.go#randomCnpjBase": { + "module": "lib/cnpj", + "start": 2700, + "end": 2947 + }, + "lib_cnpj.go#cnpjCheckDigit": { + "module": "lib/cnpj", + "start": 687, + "end": 1184 + }, + "lib_cnpj.go#hasLetter": { + "module": "lib/cnpj", + "start": 1706, + "end": 2363 + }, + "lib_cnpj.go#hasValidCnpjChecksum": { + "module": "lib/cnpj", + "start": 1265, + "end": 1478 + }, + "lib_cnpj.go#isRepeatedCnpj": { + "module": "lib/cnpj", + "start": 3028, + "end": 3247 + }, + "generate-cnpj.go#GenerateCnpj": { + "module": "generate-cnpj", + "start": 982, + "end": 1571 + }, + "lib_cpf.go#randomCpfBase": { + "module": "lib/cpf", + "start": 999, + "end": 1196 + }, + "lib_cpf.go#cpfCheckDigit": { + "module": "lib/cpf", + "start": 394, + "end": 687 + }, + "lib_cpf.go#cpfCheckDigit1": { + "module": "lib/cpf", + "start": 394, + "end": 687 + }, + "lib_cpf.go#isRepeated": { + "module": "lib/cpf", + "start": 1283, + "end": 1499 + }, + "generate-cpf.go#GenerateCpf": { + "module": "generate-cpf", + "start": 1014, + "end": 1566 + }, + "get-address-info-by-cep.go#getWithRetry": { + "module": "get-address-info-by-cep", + "start": 1086, + "end": 1491 + }, + "get-address-info-by-cep.go#isOk": { + "module": "get-address-info-by-cep", + "start": 1529, + "end": 1622 + }, + "get-address-info-by-cep.go#fetchViaCep": { + "module": "get-address-info-by-cep", + "start": 1701, + "end": 2300 + }, + "get-address-info-by-cep.go#fetchBrasilApi": { + "module": "get-address-info-by-cep", + "start": 2351, + "end": 2957 + }, + "get-address-info-by-cep.go#GetAddressInfoByCep": { + "module": "get-address-info-by-cep", + "start": 3382, + "end": 3789 + }, + "lib_json.go#matchesAt": { + "module": "lib/json", + "start": 582, + "end": 1320 + }, + "lib_json.go#isSpace": { + "module": "lib/json", + "start": 1370, + "end": 1497 + }, + "lib_json.go#hexValue": { + "module": "lib/json", + "start": 1568, + "end": 2072 + }, + "lib_json.go#jsonStringField": { + "module": "lib/json", + "start": 2300, + "end": 4239 + }, + "lib_civil.go#civilDate": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "lib_civil.go#civilDate1": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "lib_civil.go#civilDate2": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "lib_civil.go#civilDate3": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "lib_civil.go#civilDate4": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "lib_civil.go#civilDate5": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "lib_civil.go#civilDate6": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "lib_civil.go#civilDate7": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "lib_civil.go#civilDate8": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "lib_civil.go#civilDate9": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "lib_civil.go#civilDate10": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "lib_easter.go#easterDayOfMarch": { + "module": "lib/easter", + "start": 527, + "end": 1128 + }, + "lib_easter.go#easterSunday": { + "module": "lib/easter", + "start": 1186, + "end": 1392 + }, + "get-holidays.go#GetHolidays": { + "module": "get-holidays", + "start": 898, + "end": 2426 + }, + "is-business-day.go#IsBusinessDay": { + "module": "is-business-day", + "start": 578, + "end": 1064 + }, + "is-valid-cnpj.go#IsValidCnpj": { + "module": "is-valid-cnpj", + "start": 1397, + "end": 2445 + }, + "is-valid-cpf.go#IsValidCpf": { + "module": "is-valid-cpf", + "start": 793, + "end": 1143 + }, + "std_date.go#floorDiv1": { + "module": "std/date", + "start": 432, + "end": 655 + }, + "std_date.go#daysFromCivil": { + "module": "std/date", + "start": 752, + "end": 1560 + }, + "std_date.go#floorDiv": { + "module": "std/date", + "start": 432, + "end": 655 + }, + "std_date.go#yearFromDays": { + "module": "std/date", + "start": 1627, + "end": 2212 + }, + "std_date.go#monthFromDays": { + "module": "std/date", + "start": 2280, + "end": 2780 + }, + "std_date.go#dayFromDays": { + "module": "std/date", + "start": 2855, + "end": 3346 + }, + "std_date.go#ymdToDays": { + "module": "std/date", + "start": 3922, + "end": 4284 + } +} diff --git a/core/out/go/cmd/driver/main.go b/core/out/go/cmd/driver/main.go new file mode 100644 index 000000000..4228dc1ca --- /dev/null +++ b/core/out/go/cmd/driver/main.go @@ -0,0 +1,163 @@ +// Code generated by the logic engine. DO NOT EDIT. +// source: _driver + +package main + +import ( + "bufio" + "encoding/json" + "fmt" + "os" + "time" + + "coreout" +) + +type request struct { + Fn string `json:"fn"` + Args []interface{} `json:"args"` +} + +func dispatch(name string, args []interface{}) (interface{}, error) { + switch name { + case "format-cnpj::formatCnpj": + return core.FormatCnpj(args[0].(string), core.FormatCnpjOptions{Pad: args[1].(map[string]interface{})["pad"].(bool), Version: args[1].(map[string]interface{})["version"].(string), Obfuscate: args[1].(map[string]interface{})["obfuscate"].(bool)}), nil + case "format-currency::formatCurrency": + return core.FormatCurrency(int(args[0].(float64)), args[1].(bool)), nil + case "generate-cnpj::generateCnpj": + return core.GenerateCnpj(environment), nil + case "generate-cpf::generateCpf": + return core.GenerateCpf(environment), nil + case "get-address-info-by-cep::getAddressInfoByCep": + value, err := core.GetAddressInfoByCep(args[0].(string), environment) + if err != nil { + return nil, err + } + return value, nil + case "get-holidays::getHolidays": + return core.GetHolidays(int(args[0].(float64))), nil + case "is-business-day::isBusinessDay": + return core.IsBusinessDay(int(args[0].(float64)), args[1].(bool)), nil + case "is-valid-cnpj::isValidCnpj": + return core.IsValidCnpj(args[0].(string), args[1].(string)), nil + case "is-valid-cpf::isValidCpf": + return core.IsValidCpf(args[0].(string)), nil + } + return nil, fmt.Errorf("unknown function %s", name) +} + +// pcg32 is the reference PCG32: same constants and default seed as the interpreter's, so +// a draw matches the reference bit for bit. A fresh instance is built for every request, +// the same way the reference model starts a fresh interpreter -- and so a fresh generator +// -- per case. +type pcg32 struct { + state uint64 + increment uint64 +} + +func newPcg32(seed uint64) *pcg32 { + p := &pcg32{increment: 1442695040888963407} + p.next() + p.state += seed + p.next() + return p +} + +func (p *pcg32) next() int { + previous := p.state + p.state = previous*6364136223846793005 + p.increment + xorshifted := uint32(((previous >> 18) ^ previous) >> 27) + rotation := uint32(previous >> 59) + return int((xorshifted >> rotation) | (xorshifted << ((-rotation) & 31))) +} + +// defaultSeed is the interpreter's own default: its constructor falls back to this seed +// whenever Capabilities.seed is left unset, which is how every conformance case runs it. +const defaultSeed uint64 = 0x853c49e6748fea9b + +// fakeCapabilities is the capability fake the differential harness drives: responses +// come from fixtures.json, a URL that is missing models a transport error, and the +// scripted latency is what decides a race. +type fixture struct { + Status int `json:"status"` + Body string `json:"body"` + LatencyMillis int `json:"latencyMillis"` +} + +type fakeCapabilities struct { + fixtures map[string]fixture + random *pcg32 +} + +func (f fakeCapabilities) Request(request core.HttpRequest) *core.HttpResponse { + answer, ok := f.fixtures[request.Url] + if !ok { + return nil + } + time.Sleep(time.Duration(answer.LatencyMillis) * time.Millisecond) + return &core.HttpResponse{Status: answer.Status, Headers: []core.HttpHeader{}, Body: answer.Body} +} + +func (f fakeCapabilities) Now() int { return 0 } + +func (f fakeCapabilities) Sleep(milliseconds int) { + time.Sleep(time.Duration(milliseconds) * time.Millisecond) +} + +func (f fakeCapabilities) NextU32() int { return f.random.next() } + +// A missing fixture file leaves every URL unanswered, which is a transport error. +var fixtures map[string]fixture = loadFixtures() + +var environment core.Capabilities + +// newEnvironment builds a fresh capability fake, so NextU32 starts from the same state +// the reference model's fresh interpreter starts from for every case. +func newEnvironment() core.Capabilities { + return fakeCapabilities{fixtures: fixtures, random: newPcg32(defaultSeed)} +} + +func loadFixtures() map[string]fixture { + fixtures := map[string]fixture{} + raw, err := os.ReadFile("fixtures.json") + if err == nil { + if err := json.Unmarshal(raw, &fixtures); err != nil { + panic(err) + } + } + return fixtures +} + +func main() { + scanner := bufio.NewScanner(os.Stdin) + scanner.Buffer(make([]byte, 1024*1024), 1024*1024) + + for scanner.Scan() { + if scanner.Text() == "" { + continue + } + + var parsed request + if err := json.Unmarshal(scanner.Bytes(), &parsed); err != nil { + panic(err) + } + + // A fresh environment per line: NextU32 starts from the same state the reference + // model's fresh interpreter starts from for every case. + environment = newEnvironment() + + value, err := dispatch(parsed.Fn, parsed.Args) + if err != nil { + out, _ := json.Marshal(map[string]interface{}{"ok": false, "error": errorName(err)}) + fmt.Println(string(out)) + continue + } + + out, _ := json.Marshal(map[string]interface{}{"ok": true, "value": value}) + fmt.Println(string(out)) + } +} + +func errorName(err error) string { + return fmt.Sprintf("%T", err)[len("*core."):] +} diff --git a/core/out/go/errors.go b/core/out/go/errors.go new file mode 100644 index 000000000..645520e61 --- /dev/null +++ b/core/out/go/errors.go @@ -0,0 +1,46 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: errors + +package core + +import "errors" + +// ErrDomain is the root every domain error wraps, so errors.Is recognizes the family. +var ErrDomain = errors.New("domain error") + +// HttpError Raised by the engine's own intrinsics. +type HttpError struct { + Message string +} + +func (e *HttpError) Error() string { return e.Message } + +func (e *HttpError) Unwrap() error { return ErrDomain } + +// GetAddressInfoByCepError Base of every error this utility raises. +type GetAddressInfoByCepError struct { + Message string +} + +func (e *GetAddressInfoByCepError) Error() string { return e.Message } + +func (e *GetAddressInfoByCepError) Unwrap() error { return ErrDomain } + +// GetAddressInfoByCepValidationError The value given is not a CEP. +type GetAddressInfoByCepValidationError struct { + Message string +} + +func (e *GetAddressInfoByCepValidationError) Error() string { return e.Message } + +func (e *GetAddressInfoByCepValidationError) Unwrap() error { return ErrDomain } + +// GetAddressInfoByCepNotFoundError No CEP service knows this CEP, or none answered. +type GetAddressInfoByCepNotFoundError struct { + Message string +} + +func (e *GetAddressInfoByCepNotFoundError) Error() string { return e.Message } + +func (e *GetAddressInfoByCepNotFoundError) Unwrap() error { return ErrDomain } diff --git a/core/out/go/fixtures.json b/core/out/go/fixtures.json new file mode 100644 index 000000000..f05926010 --- /dev/null +++ b/core/out/go/fixtures.json @@ -0,0 +1,27 @@ +{ + "https://viacep.com.br/ws/01310100/json/": { + "status": 200, + "body": "{\"cep\":\"01310-100\",\"logradouro\":\"Avenida Paulista\",\"bairro\":\"Bela Vista\",\"localidade\":\"São Paulo\",\"uf\":\"SP\"}", + "latencyMillis": 60 + }, + "https://brasilapi.com.br/api/cep/v1/01310100": { + "status": 200, + "body": "{\"cep\":\"01310100\",\"state\":\"SP\",\"city\":\"São Paulo\",\"neighborhood\":\"Bela Vista\",\"street\":\"Avenida Paulista\"}", + "latencyMillis": 20 + }, + "https://brasilapi.com.br/api/cep/v1/30130010": { + "status": 200, + "body": "{\"cep\":\"30130010\",\"state\":\"MG\",\"city\":\"Belo Horizonte\",\"neighborhood\":\"Centro\",\"street\":\"Avenida Afonso Pena\"}", + "latencyMillis": 40 + }, + "https://viacep.com.br/ws/99999999/json/": { + "status": 200, + "body": "{\"erro\":true}", + "latencyMillis": 10 + }, + "https://brasilapi.com.br/api/cep/v1/99999999": { + "status": 404, + "body": "{\"message\":\"not found\"}", + "latencyMillis": 10 + } +} diff --git a/core/out/go/format-cnpj.go b/core/out/go/format-cnpj.go new file mode 100644 index 000000000..10cea0691 --- /dev/null +++ b/core/out/go/format-cnpj.go @@ -0,0 +1,33 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: format-cnpj +// content: 7541034a1f26 + +package core + +type FormatCnpjOptions struct { + Pad bool + Version string + Obfuscate bool +} + +// Formats a CNPJ value as `00.000.000/0000-00`. +// +// The core takes a string and a fully normalized options record; reading a number, a missing +// options object or a truthy non-boolean is the DX's job. +func FormatCnpj(value string, options FormatCnpjOptions) string { + tmp1 := "" + if options.Version == "2" { + tmp1 = keepAlphanumeric(value) + } else { + tmp1 = keepDigits(value) + } + sanitized := tmp1 + tmp2 := "" + if options.Obfuscate { + tmp2 = "**.000.000/0000-**" + } else { + tmp2 = "00.000.000/0000-00" + } + return formatWithPattern(sanitized, tmp2, options.Pad) +} diff --git a/core/out/go/format-currency.go b/core/out/go/format-currency.go new file mode 100644 index 000000000..9e0770bf9 --- /dev/null +++ b/core/out/go/format-currency.go @@ -0,0 +1,49 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: format-currency +// content: 1e0947bdf328 + +package core + +import ( + "strconv" +) + +// Formats an exact amount in Brazilian Real, with two decimal places. +// +// The separators are the ones Lei nº 9.069/1995 art. 1º prescribes and the CLDR pt-BR data uses: +// `.` between thousands, `,` before the centavos, and a non-breaking space after `R$`. A negative +// amount puts the sign before the symbol, `-R$ 10,50`, the shape `Intl.NumberFormat` produces. +// +// Turning a host value into an exact amount is the DX's job, and so is the rounding that +// conversion needs; see docs/contracts.md, which records exactly how the published package rounds. +func FormatCurrency(value int, symbol bool) string { + negative := value < 0 + unscaled := absInt(value) + digits := padStart(strconv.Itoa(unscaled), 3, "0") + cut := max((len(digits) - 2), 0) + whole := digits[0:cut] + cents := digits[cut:len(digits)] + body := ((groupThousands(keepDigits(whole)) + ",") + cents) + tmp1 := "" + if symbol { + tmp1 = ("R$" + string(func() []byte { + __pts := []int{32} + __bs := make([]byte, len(__pts)) + for __i, __p := range __pts { + __bs[__i] = byte(__p) + } + return __bs + }())) + } else { + tmp1 = "" + } + prefix := tmp1 + tmp2 := "" + if negative { + tmp2 = (("-" + prefix) + body) + } else { + tmp2 = (prefix + body) + } + return tmp2 +} diff --git a/core/out/go/generate-cnpj.go b/core/out/go/generate-cnpj.go new file mode 100644 index 000000000..5d472bbb8 --- /dev/null +++ b/core/out/go/generate-cnpj.go @@ -0,0 +1,30 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: generate-cnpj +// content: 41b441a54f18 + +package core + +import ( + "strconv" +) + +// Generates a valid random CNPJ (Cadastro Nacional da Pessoa Jurídica) in the numeric format: 14 +// digits, under the check digit rule both CNPJ versions share. +// +// Matches the published `generateCnpj()` called with no options: a random 8-digit root and +// 4-digit branch (the "número de ordem"), redrawn while every digit of the 12-digit base is the +// same, followed by its two check digits. The alphanumeric version and a chosen branch are DX +// concerns layered on the same base and check digit rule, not a different generator. +func GenerateCnpj(env Capabilities) string { + base := randomCnpjBase(env) + for attempt := 0; attempt < 8; attempt++ { + if !isRepeatedRun(base) { + break + } + base = randomCnpjBase(env) + } + firstDigit := strconv.Itoa(cnpjCheckDigit((base + "00"), []int{5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2})) + secondDigit := strconv.Itoa(cnpjCheckDigit(((base + firstDigit) + "0"), []int{6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2})) + return ((base + firstDigit) + secondDigit) +} diff --git a/core/out/go/generate-cpf.go b/core/out/go/generate-cpf.go new file mode 100644 index 000000000..40ada7f0d --- /dev/null +++ b/core/out/go/generate-cpf.go @@ -0,0 +1,31 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: generate-cpf +// content: 00fc978e8350 + +package core + +import ( + "strconv" +) + +// Generates a valid random CPF (Cadastro de Pessoas Físicas): 11 digits, under the check digit +// rule (weights 10..2 and 11..2) the Receita Federal's Manual de Preenchimento da e-Financeira, +// Anexo II specifies. +// +// Matches the published `generateCpf()` called with no state: a random 9-digit base — 8 digits +// plus a região fiscal digit, also drawn at random here — redrawn while every digit of it is the +// same, followed by its two check digits. The state code option is a DX concern: it only ever +// picks which digit the 9th position draws from, never how the rest of the document is built. +func GenerateCpf(env Capabilities) string { + base := randomCpfBase(env) + for attempt := 0; attempt < 8; attempt++ { + if !isRepeatedRun(base) { + break + } + base = randomCpfBase(env) + } + firstDigit := strconv.Itoa(cpfCheckDigit((base + "00"))) + secondDigit := strconv.Itoa(cpfCheckDigit1(((base + firstDigit) + "0"))) + return ((base + firstDigit) + secondDigit) +} diff --git a/core/out/go/get-address-info-by-cep.go b/core/out/go/get-address-info-by-cep.go new file mode 100644 index 000000000..0fcac6265 --- /dev/null +++ b/core/out/go/get-address-info-by-cep.go @@ -0,0 +1,86 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: get-address-info-by-cep +// content: daa7ccc2e0b6 + +package core + +import ( + "regexp" +) + +var getAddressInfoByCepPattern1 = regexp.MustCompile("\\A[0-9]{8}\\z") + +type AddressInfo struct { + Cep string + State string + City string + Neighborhood string + Street string +} + +// One GET, retried the way the published package retries: twice more, 250 ms apart. +func getWithRetry(url string, env Capabilities) *HttpResponse { + for attempt := 0; attempt < 3; attempt++ { + if attempt > 0 { + env.Sleep(250) + } + response := env.Request(HttpRequest{Method: "GET", Url: url, Headers: []HttpHeader{}, Body: "", TimeoutMillis: 10000}) + if response != nil { + return response + } + } + return nil +} + +// Whether the status is a 2xx. +func isOk(status int) bool { + return ((status >= 200) && (status < 300)) +} + +// ViaCEP answers a JSON object, and marks an unknown CEP with `"erro"`. +func fetchViaCep(cep string, env Capabilities) *AddressInfo { + response := getWithRetry((("https://viacep.com.br/ws/" + cep) + "/json/"), env) + if (response == nil) || !isOk((*response).Status) { + return nil + } + code := orElse(jsonStringField((*response).Body, "cep"), "") + if code == "" { + return nil + } + return ptr(AddressInfo{Cep: keepDigits(code), State: orElse(jsonStringField((*response).Body, "uf"), ""), City: orElse(jsonStringField((*response).Body, "localidade"), ""), Neighborhood: orElse(jsonStringField((*response).Body, "bairro"), ""), Street: orElse(jsonStringField((*response).Body, "logradouro"), "")}) +} + +// BrasilAPI answers 404 for an unknown CEP. +func fetchBrasilApi(cep string, env Capabilities) *AddressInfo { + response := getWithRetry(("https://brasilapi.com.br/api/cep/v1/" + cep), env) + if (response == nil) || !isOk((*response).Status) { + return nil + } + code := orElse(jsonStringField((*response).Body, "cep"), "") + if code == "" { + return nil + } + return ptr(AddressInfo{Cep: keepDigits(code), State: orElse(jsonStringField((*response).Body, "state"), ""), City: orElse(jsonStringField((*response).Body, "city"), ""), Neighborhood: orElse(jsonStringField((*response).Body, "neighborhood"), ""), Street: orElse(jsonStringField((*response).Body, "street"), "")}) +} + +// The address of a CEP, from the first service that answers. +// +// The two services are queried concurrently and the first answer wins; the losing request may +// still finish, and its answer is dropped, which is why only idempotent GETs belong here. Each +// request is retried twice, 250 ms apart, exactly as the published package does. Turning a host +// value into the 8 digits this takes is the DX's job. +func GetAddressInfoByCep(cep string, env Capabilities) (AddressInfo, error) { + if !(getAddressInfoByCepPattern1.MatchString(cep)) { + return AddressInfo{}, &GetAddressInfoByCepValidationError{Message: "CEP inv\u00e1lido"} + } + address := raceFirstSome([]func() *AddressInfo{func() *AddressInfo { + return fetchViaCep(cep, env) + }, func() *AddressInfo { + return fetchBrasilApi(cep, env) + }}) + if address == nil { + return AddressInfo{}, &GetAddressInfoByCepNotFoundError{Message: "CEP n\u00e3o encontrado"} + } + return (*address), nil +} diff --git a/core/out/go/get-holidays.go b/core/out/go/get-holidays.go new file mode 100644 index 000000000..f2e471835 --- /dev/null +++ b/core/out/go/get-holidays.go @@ -0,0 +1,40 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: get-holidays +// content: b4d687034433 + +package core + +type Holiday struct { + Name string + Date int + Type string +} + +// The Brazilian national holidays of a year, sorted by date. +// +// The order is the one the published package produces: the fixed holidays in statutory order, +// then the Easter-derived ones, sorted by date with a stable sort, so two holidays on the same +// day keep the order they were built in. State holidays are not part of this pilot. +func GetHolidays(year int) []Holiday { + holidays := []Holiday{} + holidays = append(holidays, Holiday{Name: "Ano novo", Date: civilDate(year), Type: "national"}) + holidays = append(holidays, Holiday{Name: "Tiradentes", Date: civilDate1(year), Type: "national"}) + holidays = append(holidays, Holiday{Name: "Dia do trabalhador", Date: civilDate2(year), Type: "national"}) + holidays = append(holidays, Holiday{Name: "Independ\u00eancia do Brasil", Date: civilDate3(year), Type: "national"}) + holidays = append(holidays, Holiday{Name: "Nossa Senhora Aparecida", Date: civilDate4(year), Type: "national"}) + holidays = append(holidays, Holiday{Name: "Finados", Date: civilDate5(year), Type: "national"}) + holidays = append(holidays, Holiday{Name: "Proclama\u00e7\u00e3o da Rep\u00fablica", Date: civilDate6(year), Type: "national"}) + holidays = append(holidays, Holiday{Name: "Natal", Date: civilDate7(year), Type: "national"}) + if year >= 2024 { + holidays = append(holidays, Holiday{Name: "Dia da Consci\u00eancia Negra", Date: civilDate8(year), Type: "national"}) + } + easter := easterSunday(year) + holidays = append(holidays, Holiday{Name: "Carnaval (ter\u00e7a-feira)", Date: orElse(dateAddDays(easter, -47), easter), Type: "optional"}) + holidays = append(holidays, Holiday{Name: "Sexta-feira Santa", Date: orElse(dateAddDays(easter, -2), easter), Type: "national"}) + holidays = append(holidays, Holiday{Name: "P\u00e1scoa", Date: easter, Type: "religious"}) + holidays = append(holidays, Holiday{Name: "Corpus Christi", Date: orElse(dateAddDays(easter, 60), easter), Type: "optional"}) + return sortedStableBy(holidays, func(holiday Holiday) int { + return holiday.Date + }, func(a, b int) int { return a - b }) +} diff --git a/core/out/go/go.mod b/core/out/go/go.mod new file mode 100644 index 000000000..d761996f9 --- /dev/null +++ b/core/out/go/go.mod @@ -0,0 +1,3 @@ +module coreout + +go 1.21 diff --git a/core/out/go/is-business-day.go b/core/out/go/is-business-day.go new file mode 100644 index 000000000..561af8ebb --- /dev/null +++ b/core/out/go/is-business-day.go @@ -0,0 +1,34 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: is-business-day +// content: 3605fae387a8 + +package core + +// Whether a date is a Brazilian business day (dia útil). +// +// A day is not a business day when it falls on a weekend, or when it is one of the holidays +// `getHolidays` lists for its year. `includeOptional` decides whether the ponto facultativo +// entries (Carnaval, Corpus Christi) count; the published package defaults it to `true`, and +// supplying that default is the DX's job. +// +// Only the years 1900 to 2099 are supported, the range the holiday rules are stated for. +func IsBusinessDay(value int, includeOptional bool) bool { + year := yearFromDays(value) + if (year < 1900) || (year > 2099) { + return false + } + weekday := (((value+3)%7+7)%7 + 1) + if (weekday == 6) || (weekday == 7) { + return false + } + for _, holiday := range GetHolidays(year) { + if !includeOptional && (holiday.Type == "optional") { + continue + } + if compareInts(holiday.Date, value) == 0 { + return false + } + } + return true +} diff --git a/core/out/go/is-valid-cnpj.go b/core/out/go/is-valid-cnpj.go new file mode 100644 index 000000000..25d102108 --- /dev/null +++ b/core/out/go/is-valid-cnpj.go @@ -0,0 +1,40 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: is-valid-cnpj +// content: 611130f5f12f + +package core + +import ( + "regexp" + "strings" +) + +var isValidCnpjPattern1 = regexp.MustCompile("\\A[0-9A-Z]{2}[\\x{9}-\\x{d} \\--/\\x{a0}\\x{1680}\\x{2000}-\\x{200a}\\x{2028}-\\x{2029}\\x{202f}\\x{205f}\\x{3000}\\x{feff}]*[0-9A-Z]{3}[\\x{9}-\\x{d} \\--/\\x{a0}\\x{1680}\\x{2000}-\\x{200a}\\x{2028}-\\x{2029}\\x{202f}\\x{205f}\\x{3000}\\x{feff}]*[0-9A-Z]{3}[\\x{9}-\\x{d} \\--/\\x{a0}\\x{1680}\\x{2000}-\\x{200a}\\x{2028}-\\x{2029}\\x{202f}\\x{205f}\\x{3000}\\x{feff}]*[0-9A-Z]{4}[\\x{9}-\\x{d} \\--/\\x{a0}\\x{1680}\\x{2000}-\\x{200a}\\x{2028}-\\x{2029}\\x{202f}\\x{205f}\\x{3000}\\x{feff}]*[0-9]{2}\\z") + +var isValidCnpjPattern2 = regexp.MustCompile("\\A[0-9]{2}[\\x{9}-\\x{d} \\--/\\x{a0}\\x{1680}\\x{2000}-\\x{200a}\\x{2028}-\\x{2029}\\x{202f}\\x{205f}\\x{3000}\\x{feff}]*[0-9]{3}[\\x{9}-\\x{d} \\--/\\x{a0}\\x{1680}\\x{2000}-\\x{200a}\\x{2028}-\\x{2029}\\x{202f}\\x{205f}\\x{3000}\\x{feff}]*[0-9]{3}[\\x{9}-\\x{d} \\--/\\x{a0}\\x{1680}\\x{2000}-\\x{200a}\\x{2028}-\\x{2029}\\x{202f}\\x{205f}\\x{3000}\\x{feff}]*[0-9]{4}[\\x{9}-\\x{d} \\--/\\x{a0}\\x{1680}\\x{2000}-\\x{200a}\\x{2028}-\\x{2029}\\x{202f}\\x{205f}\\x{3000}\\x{feff}]*[0-9]{2}\\z") + +// Validates a CNPJ (Cadastro Nacional da Pessoa Jurídica), numeric or alphanumeric. +// +// Version `"2"` accepts the alphanumeric format as well; a value with no letters is always read +// as the numeric one, which is also where the reserved repeated numbers are rejected. Mapping a +// missing or unexpected `options.version` onto `"1"` is the DX's job. +func IsValidCnpj(cnpj string, version string) bool { + trimmed := strings.Trim(cnpj, "\t\n\u000b\u000c\r \u00a0\u1680\u2000\u2001\u2002\u2003\u2004\u2005\u2006\u2007\u2008\u2009\u200a\u2028\u2029\u202f\u205f\u3000\ufeff") + if version == "2" { + cleaned := keepAlphanumeric(cnpj) + if hasLetter(cleaned) && (len(cleaned) == 14) { + return (isValidCnpjPattern1.MatchString(strings.Map(func(scalar rune) rune { + if scalar >= 'a' && scalar <= 'z' { + return scalar - 32 + } + return scalar + }, trimmed)) && hasValidCnpjChecksum(cleaned)) + } + } + numeric := keepDigits(cnpj) + if len(numeric) != 14 { + return false + } + return ((isValidCnpjPattern2.MatchString(trimmed) && !isRepeatedCnpj(numeric)) && hasValidCnpjChecksum(numeric)) +} diff --git a/core/out/go/is-valid-cpf.go b/core/out/go/is-valid-cpf.go new file mode 100644 index 000000000..ba770ffd6 --- /dev/null +++ b/core/out/go/is-valid-cpf.go @@ -0,0 +1,31 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: is-valid-cpf +// content: dae364e11fb3 + +package core + +import ( + "regexp" + "strings" +) + +var isValidCpfPattern1 = regexp.MustCompile("\\A[0-9]{3}[\\x{9}-\\x{d} \\--/\\x{a0}\\x{1680}\\x{2000}-\\x{200a}\\x{2028}-\\x{2029}\\x{202f}\\x{205f}\\x{3000}\\x{feff}]*[0-9]{3}[\\x{9}-\\x{d} \\--/\\x{a0}\\x{1680}\\x{2000}-\\x{200a}\\x{2028}-\\x{2029}\\x{202f}\\x{205f}\\x{3000}\\x{feff}]*[0-9]{3}[\\x{9}-\\x{d} \\--/\\x{a0}\\x{1680}\\x{2000}-\\x{200a}\\x{2028}-\\x{2029}\\x{202f}\\x{205f}\\x{3000}\\x{feff}]*[0-9]{2}\\z") + +// Validates a CPF (Cadastro de Pessoas Físicas). +// +// The core takes the value as written, accepting the usual mask characters; turning a host value +// into a string is the DX's job. +func IsValidCpf(cpf string) bool { + if !(isValidCpfPattern1.MatchString(strings.Trim(cpf, "\t\n\u000b\u000c\r \u00a0\u1680\u2000\u2001\u2002\u2003\u2004\u2005\u2006\u2007\u2008\u2009\u200a\u2028\u2029\u202f\u205f\u3000\ufeff"))) { + return false + } + digits := keepDigits(cpf) + if len(digits) != 11 { + return false + } + if isRepeated(digits) { + return false + } + return ((digitAt2(digits) == cpfCheckDigit(digits)) && (digitAt3(digits) == cpfCheckDigit1(digits))) +} diff --git a/core/out/go/lib_civil.go b/core/out/go/lib_civil.go new file mode 100644 index 000000000..172c6c05e --- /dev/null +++ b/core/out/go/lib_civil.go @@ -0,0 +1,61 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/civil +// content: c2aca1eb50dc + +package core + +// A fixed day of a year, with the unreachable fallback named once. +func civilDate(year int) int { + return orElse(ymdToDays(year, 1, 1), min(max(0, -719162), 2932896)) +} + +// A fixed day of a year, with the unreachable fallback named once. +func civilDate1(year int) int { + return orElse(ymdToDays(year, 4, 21), min(max(0, -719162), 2932896)) +} + +// A fixed day of a year, with the unreachable fallback named once. +func civilDate2(year int) int { + return orElse(ymdToDays(year, 5, 1), min(max(0, -719162), 2932896)) +} + +// A fixed day of a year, with the unreachable fallback named once. +func civilDate3(year int) int { + return orElse(ymdToDays(year, 9, 7), min(max(0, -719162), 2932896)) +} + +// A fixed day of a year, with the unreachable fallback named once. +func civilDate4(year int) int { + return orElse(ymdToDays(year, 10, 12), min(max(0, -719162), 2932896)) +} + +// A fixed day of a year, with the unreachable fallback named once. +func civilDate5(year int) int { + return orElse(ymdToDays(year, 11, 2), min(max(0, -719162), 2932896)) +} + +// A fixed day of a year, with the unreachable fallback named once. +func civilDate6(year int) int { + return orElse(ymdToDays(year, 11, 15), min(max(0, -719162), 2932896)) +} + +// A fixed day of a year, with the unreachable fallback named once. +func civilDate7(year int) int { + return orElse(ymdToDays(year, 12, 25), min(max(0, -719162), 2932896)) +} + +// A fixed day of a year, with the unreachable fallback named once. +func civilDate8(year int) int { + return orElse(ymdToDays(year, 11, 20), min(max(0, -719162), 2932896)) +} + +// A fixed day of a year, with the unreachable fallback named once. +func civilDate9(year int, day int) int { + return orElse(ymdToDays(year, 3, day), min(max(0, -719162), 2932896)) +} + +// A fixed day of a year, with the unreachable fallback named once. +func civilDate10(year int, day int) int { + return orElse(ymdToDays(year, 4, day), min(max(0, -719162), 2932896)) +} diff --git a/core/out/go/lib_cnpj.go b/core/out/go/lib_cnpj.go new file mode 100644 index 000000000..cb8416f0e --- /dev/null +++ b/core/out/go/lib_cnpj.go @@ -0,0 +1,64 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/cnpj +// content: 78a3fa4783b9 + +package core + +var libCnpjTable1 = []int{5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2} + +var libCnpjTable2 = []int{6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2} + +// A random numeric CNPJ base: an 8-digit root and a 4-digit branch, each digit drawn +// independently — matches the published `generateCnpj()` called with no branch, where an unset +// branch also draws those 4 digits at random. Twelve separate draws, not a loop, is what lets the +// result stay exactly 12 digits long. +func randomCnpjBase(env Capabilities) string { + return (((((((((((randomDigit(env) + randomDigit(env)) + randomDigit(env)) + randomDigit(env)) + randomDigit(env)) + randomDigit(env)) + randomDigit(env)) + randomDigit(env)) + randomDigit(env)) + randomDigit(env)) + randomDigit(env)) + randomDigit(env)) +} + +// The check digit of a CNPJ base, under the rule both versions share. +func cnpjCheckDigit(cnpj string, weights []int) int { + sum := 0 + for index := 0; index < len(weights); index++ { + sum = (sum + ((int(cnpj[index]) - 48) * weights[index])) + } + remainder := (sum % 11) + tmp1 := 0 + if remainder < 2 { + tmp1 = 0 + } else { + tmp1 = (11 - remainder) + } + return tmp1 +} + +// Whether the value holds at least one upper cased ASCII letter. +// +// The scan reads positions rather than materializing the scalars, which the checked accessor +// makes safe without a proof about the length. +func hasLetter(value string) bool { + for index := 0; index < len(value); index++ { + point := orElse(codeAt(value, index), 0) + if (point >= 65) && (point <= 90) { + return true + } + } + return false +} + +// Whether both check digits of a 14 character CNPJ match its base. +func hasValidCnpjChecksum(cnpj string) bool { + return (((int(cnpj[12]) - 48) == cnpjCheckDigit(cnpj, libCnpjTable1)) && ((int(cnpj[13]) - 48) == cnpjCheckDigit(cnpj, libCnpjTable2))) +} + +// Whether every character of a 14 character value is the same one. +func isRepeatedCnpj(value string) bool { + first := int(value[0]) + for index := 1; index < 14; index++ { + if int(value[index]) != first { + return false + } + } + return true +} diff --git a/core/out/go/lib_cpf.go b/core/out/go/lib_cpf.go new file mode 100644 index 000000000..58f36dbf8 --- /dev/null +++ b/core/out/go/lib_cpf.go @@ -0,0 +1,56 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/cpf +// content: 5b319831cfe4 + +package core + +// A random CPF base: 8 digits plus a região fiscal digit, each drawn independently — matches the +// published `generateCpf()` called with no state, where an unset state also draws that 9th digit +// at random. Nine separate draws, not a loop, is what lets the result stay exactly 9 digits long. +func randomCpfBase(env Capabilities) string { + return ((((((((randomDigit(env) + randomDigit(env)) + randomDigit(env)) + randomDigit(env)) + randomDigit(env)) + randomDigit(env)) + randomDigit(env)) + randomDigit(env)) + randomDigit(env)) +} + +// The check digit of a CPF base, under the Receita Federal rule (weights 10..2 and 11..2). +func cpfCheckDigit(cpf string) int { + sum := 0 + for index := 0; index < 9; index++ { + sum = (sum + (digitAt(cpf, index) * (10 - index))) + } + remainder := (sum % 11) + tmp1 := 0 + if remainder < 2 { + tmp1 = 0 + } else { + tmp1 = (11 - remainder) + } + return tmp1 +} + +// The check digit of a CPF base, under the Receita Federal rule (weights 10..2 and 11..2). +func cpfCheckDigit1(cpf string) int { + sum := 0 + for index := 0; index < 10; index++ { + sum = (sum + (digitAt(cpf, index) * (11 - index))) + } + remainder := (sum % 11) + tmp1 := 0 + if remainder < 2 { + tmp1 = 0 + } else { + tmp1 = (11 - remainder) + } + return tmp1 +} + +// Whether every scalar of the value is the same one, e.g. "00000000000". +func isRepeated(value string) bool { + first := int(value[0]) + for index := 1; index < 11; index++ { + if int(value[index]) != first { + return false + } + } + return true +} diff --git a/core/out/go/lib_digits.go b/core/out/go/lib_digits.go new file mode 100644 index 000000000..a716dc90c --- /dev/null +++ b/core/out/go/lib_digits.go @@ -0,0 +1,70 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/digits +// content: 780bb464b89e + +package core + +import ( + "strings" +) + +// Keeps only the ASCII digits and letters of a value, upper casing the letters. +func keepAlphanumeric(value string) string { + return strings.ToUpper(func() string { + __value := value + __out := make([]byte, 0, len(__value)) + for __i := 0; __i < len(__value); __i++ { + __b := __value[__i] + __c := int(__b) + if (__c >= 48 && __c <= 57) || (__c >= 65 && __c <= 90) || (__c >= 97 && __c <= 122) { + __out = append(__out, __b) + } + } + return string(__out) + }()) +} + +// Keeps only the ASCII digits of a value, dropping every mask character. +func keepDigits(value string) string { + return func() string { + __value := value + __out := make([]byte, 0, len(__value)) + for __i := 0; __i < len(__value); __i++ { + __b := __value[__i] + __c := int(__b) + if __c >= 48 && __c <= 57 { + __out = append(__out, __b) + } + } + return string(__out) + }() +} + +// Whether every scalar of the value is the same one, for whatever length the caller proved — +// `isRepeated` and `isRepeatedCnpj` do the same check for one specific length; this one serves a +// generator that has to run it on a base shorter than the document it is building. +func isRepeatedRun(value string) bool { + first := int(value[0]) + for index := 1; index < len(value); index++ { + if int(value[index]) != first { + return false + } + } + return true +} + +// The numeric value of one ASCII digit. +func digitAt(value string, index int) int { + return (int(value[index]) - 48) +} + +// The numeric value of one ASCII digit. +func digitAt2(value string) int { + return (int(value[9]) - 48) +} + +// The numeric value of one ASCII digit. +func digitAt3(value string) int { + return (int(value[10]) - 48) +} diff --git a/core/out/go/lib_easter.go b/core/out/go/lib_easter.go new file mode 100644 index 000000000..dbc67c089 --- /dev/null +++ b/core/out/go/lib_easter.go @@ -0,0 +1,34 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/easter +// content: c99e5cf5c72b + +package core + +// The day of March (1 to 31) or April (32 to 56) Easter falls on, as a day-of-March offset. +func easterDayOfMarch(year int) int { + a := (year % 19) + b := (year / 100) + c := (year % 100) + d := (b / 4) + e := (b % 4) + h := ((((((19 * a) + b) - d) - 6) + 15) % 30) + i := (c / 4) + k := (c % 4) + l := (((((32 + (2 * e)) + (2 * i)) - h) - k) % 7) + m := (((a + (11 * h)) + (22 * l)) / 451) + day := (((h + l) - (7 * m)) + 114) + return min(max((((day%31)+1)+(((day/31)-3)*31)), 22), 56) +} + +// Easter Sunday of a year, as a civil date. +func easterSunday(year int) int { + dayOfMarch := easterDayOfMarch(year) + tmp1 := 0 + if dayOfMarch <= 31 { + tmp1 = civilDate9(year, dayOfMarch) + } else { + tmp1 = civilDate10(year, (dayOfMarch - 31)) + } + return tmp1 +} diff --git a/core/out/go/lib_format.go b/core/out/go/lib_format.go new file mode 100644 index 000000000..02cb9c938 --- /dev/null +++ b/core/out/go/lib_format.go @@ -0,0 +1,79 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/format +// content: 0038a81c2e4d + +package core + +// How many scalars of the value a pattern consumes. +func patternSlots(pattern string) int { + slots := 0 + for index := 0; index < len(pattern); index++ { + symbol := orElse(charAt(pattern, index), "") + if (symbol == "0") || (symbol == "*") { + slots = (slots + 1) + } + } + return slots +} + +// Formats a value against a pattern, optionally left padding it with zeros first. +func formatWithPattern(value string, pattern string, pad bool) string { + tmp1 := "" + if pad { + tmp1 = padStart(value, patternSlots(pattern), "0") + } else { + tmp1 = value + } + padded := tmp1 + out := "" + taken := 0 + for index := 0; index < len(pattern); index++ { + symbol := orElse(charAt(pattern, index), "") + if (symbol == "0") || (symbol == "*") { + if taken >= len(padded) { + return out + } + tmp2 := "" + if symbol == "*" { + tmp2 = "*" + } else { + tmp2 = orElse(charAt(padded, taken), "") + } + out = (out + tmp2) + taken = (taken + 1) + } else { + if taken < len(padded) { + out = (out + symbol) + } + } + } + return out +} + +// Groups the whole part with `.` every three digits, the pt-BR convention. +func groupThousands(whole string) string { + out := []int{} + scalars := func() []int { + __bs := []byte(whole) + __pts := make([]int, len(__bs)) + for __i, __b := range __bs { + __pts[__i] = int(__b) + } + return __pts + }() + for index := 0; index < len(scalars); index++ { + if (index > 0) && (((len(scalars) - index) % 3) == 0) { + out = append(out, 46) + } + out = append(out, orElse(at(scalars, index), 48)) + } + return string(func() []byte { + __pts := out + __bs := make([]byte, len(__pts)) + for __i, __p := range __pts { + __bs[__i] = byte(__p) + } + return __bs + }()) +} diff --git a/core/out/go/lib_json.go b/core/out/go/lib_json.go new file mode 100644 index 000000000..4baa6d48e --- /dev/null +++ b/core/out/go/lib_json.go @@ -0,0 +1,123 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/json +// content: 02cc75dfd626 + +package core + +// Whether `needle` occurs in `points` at `start`. +func matchesAt(points []int, needle []int, start int) bool { + for offset := 0; offset < len(needle); offset++ { + if orElse(at(points, (start+offset)), -1) != orElse(at(needle, offset), -2) { + return false + } + } + return true +} + +// Whether a code point is JSON whitespace. +func isSpace(point int) bool { + return ((((point == 32) || (point == 9)) || (point == 10)) || (point == 13)) +} + +// The hexadecimal value of four scalars, for a `\uXXXX` escape. +func hexValue(points []int, start int) int { + value := 0 + for offset := 0; offset < 4; offset++ { + point := orElse(at(points, (start+offset)), 48) + digit := 0 + if (point >= 48) && (point <= 57) { + digit = (point - 48) + } else { + if (point >= 97) && (point <= 102) { + digit = (point - 87) + } else { + if (point >= 65) && (point <= 70) { + digit = (point - 55) + } + } + } + value = ((value * 16) + digit) + } + return min(value, 65535) +} + +// The string value of a top-level JSON field, or absent when the field is missing or is not a +// string. Escapes are decoded; a surrogate pair is left as its two escaped halves, which no CEP +// provider emits. +func jsonStringField(body string, key string) *string { + points := codePoints(body) + needle := func() []int { + __bs := []byte((("\"" + key) + "\"")) + __pts := make([]int, len(__bs)) + for __i, __b := range __bs { + __pts[__i] = int(__b) + } + return __pts + }() + for index := 0; index < len(points); index++ { + if !matchesAt(points, needle, index) { + continue + } + cursor := (index + len(needle)) + for skip := 0; skip < 8; skip++ { + if isSpace(orElse(at(points, cursor), 0)) { + cursor = (cursor + 1) + } + } + if orElse(at(points, cursor), 0) != 58 { + continue + } + cursor = (cursor + 1) + for skip := 0; skip < 8; skip++ { + if isSpace(orElse(at(points, cursor), 0)) { + cursor = (cursor + 1) + } + } + if orElse(at(points, cursor), 0) != 34 { + continue + } + cursor = (cursor + 1) + out := []int{} + for step := 0; step < len(points); step++ { + point := orElse(at(points, cursor), -1) + if (point == -1) || (point == 34) { + return ptr(fromCodePoints(out)) + } + if point == 92 { + escaped := orElse(at(points, (cursor+1)), -1) + if escaped == 110 { + out = append(out, 10) + cursor = (cursor + 2) + } else { + if escaped == 116 { + out = append(out, 9) + cursor = (cursor + 2) + } else { + if escaped == 114 { + out = append(out, 13) + cursor = (cursor + 2) + } else { + if escaped == 117 { + out = append(out, hexValue(points, (cursor+2))) + cursor = (cursor + 6) + } else { + if escaped >= 0 { + out = append(out, escaped) + cursor = (cursor + 2) + } else { + cursor = (cursor + 1) + } + } + } + } + } + } else { + out = append(out, point) + cursor = (cursor + 1) + } + } + return ptr(fromCodePoints(out)) + } + return nil +} diff --git a/core/out/go/lib_random.go b/core/out/go/lib_random.go new file mode 100644 index 000000000..8baee8fb6 --- /dev/null +++ b/core/out/go/lib_random.go @@ -0,0 +1,30 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/random +// content: 2101f5409061 + +package core + +import ( + "strconv" +) + +// A uniform integer in `[0, bound)`, by rejection sampling rather than `% bound`: the modulo of a +// fixed-width draw is biased whenever `bound` does not divide 2^32 evenly, and that bias would +// have to match, digit for digit, across three unrelated standard libraries to stay invisible. +// Rejecting the biased tail of the draw removes it instead. +func randomBelow(env Capabilities) int { + limit := 4294967290 + for attempt := 0; attempt < 32; attempt++ { + draw := env.NextU32() + if draw < limit { + return (draw % 10) + } + } + return (env.NextU32() % 10) +} + +// One random ASCII digit. +func randomDigit(env Capabilities) string { + return strconv.Itoa(randomBelow(env)) +} diff --git a/core/out/go/std_date.go b/core/out/go/std_date.go new file mode 100644 index 000000000..b8cc2a472 --- /dev/null +++ b/core/out/go/std_date.go @@ -0,0 +1,116 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: std/date +// content: 8041c981a090 + +package core + +// Floor division, which the calendar algorithms need for negative years. +func floorDiv1(value int) int { + quotient := (value / 400) + if (value < 0) && ((quotient * 400) != value) { + return (quotient - 1) + } + return quotient +} + +// Days since 1970-01-01 for a year, month and day already known to be a real date. +func daysFromCivil(year int, month int, day int) int { + tmp1 := 0 + if month <= 2 { + tmp1 = (year - 1) + } else { + tmp1 = year + } + shifted := tmp1 + era := floorDiv1(shifted) + yearOfEra := (shifted - (era * 400)) + tmp2 := 0 + if month > 2 { + tmp2 = (month - 3) + } else { + tmp2 = (month + 9) + } + monthTerm := tmp2 + dayOfYear := (((((153 * monthTerm) + 2) / 5) + day) - 1) + dayOfEra := ((((yearOfEra * 365) + (yearOfEra / 4)) - (yearOfEra / 100)) + dayOfYear) + return min(max((((era*146097)+dayOfEra)-719468), -719162), 2932896) +} + +// Floor division, which the calendar algorithms need for negative years. +func floorDiv(value int) int { + quotient := (value / 146097) + if (value < 0) && ((quotient * 146097) != value) { + return (quotient - 1) + } + return quotient +} + +// The year of a date given as days since 1970-01-01. +func yearFromDays(days int) int { + shifted := (days + 719468) + era := floorDiv(shifted) + dayOfEra := (shifted - (era * 146097)) + yearOfEra := ((((dayOfEra - (dayOfEra / 1460)) + (dayOfEra / 36524)) - (dayOfEra / 146096)) / 365) + year := (yearOfEra + (era * 400)) + dayOfYear := (dayOfEra - (((365 * yearOfEra) + (yearOfEra / 4)) - (yearOfEra / 100))) + monthPrime := (((5 * dayOfYear) + 2) / 153) + tmp1 := 0 + if monthPrime < 10 { + tmp1 = (monthPrime + 3) + } else { + tmp1 = (monthPrime - 9) + } + month := tmp1 + tmp2 := 0 + if month <= 2 { + tmp2 = (year + 1) + } else { + tmp2 = year + } + return min(max(tmp2, 1), 9999) +} + +// The month of a date given as days since 1970-01-01. +func monthFromDays(days int) int { + shifted := (days + 719468) + era := floorDiv(shifted) + dayOfEra := (shifted - (era * 146097)) + yearOfEra := ((((dayOfEra - (dayOfEra / 1460)) + (dayOfEra / 36524)) - (dayOfEra / 146096)) / 365) + dayOfYear := (dayOfEra - (((365 * yearOfEra) + (yearOfEra / 4)) - (yearOfEra / 100))) + monthPrime := (((5 * dayOfYear) + 2) / 153) + tmp1 := 0 + if monthPrime < 10 { + tmp1 = (monthPrime + 3) + } else { + tmp1 = (monthPrime - 9) + } + return min(max(tmp1, 1), 12) +} + +// The day of month of a date given as days since 1970-01-01. +func dayFromDays(days int) int { + shifted := (days + 719468) + era := floorDiv(shifted) + dayOfEra := (shifted - (era * 146097)) + yearOfEra := ((((dayOfEra - (dayOfEra / 1460)) + (dayOfEra / 36524)) - (dayOfEra / 146096)) / 365) + dayOfYear := (dayOfEra - (((365 * yearOfEra) + (yearOfEra / 4)) - (yearOfEra / 100))) + monthPrime := (((5 * dayOfYear) + 2) / 153) + return min(max(((dayOfYear-(((153*monthPrime)+2)/5))+1), 1), 31) +} + +// Days since 1970-01-01, or absent when the components do not name a real date. +// +// The bounds are checked here rather than in a helper because the checker reads a guard, not a +// called predicate: after this `if`, the three components carry the ranges `daysFromCivil` +// requires, and the round trip rejects a day the month does not have. +func ymdToDays(year int, month int, day int) *int { + if (((((year < 1) || (year > 9999)) || (month < 1)) || (month > 12)) || (day < 1)) || (day > 31) { + return nil + } + days := daysFromCivil(year, month, day) + if ((yearFromDays(days) != year) || (monthFromDays(days) != month)) || (dayFromDays(days) != day) { + return nil + } + return ptr(days) +} diff --git a/core/out/go/support.go b/core/out/go/support.go new file mode 100644 index 000000000..edaba1977 --- /dev/null +++ b/core/out/go/support.go @@ -0,0 +1,267 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: support + +package core + +import ( + "slices" +) + +// ptr wraps a present value, which is how an Option is represented in Go. +func ptr[T any](value T) *T { return &value } + +// orElse answers the value, or the fallback when the option is absent. +func orElse[T any](value *T, fallback T) T { + if value == nil { + return fallback + } + return *value +} + +func codePoints(value string) []int { + points := make([]int, 0, len(value)) + for _, scalar := range value { + points = append(points, int(scalar)) + } + return points +} + +func fromCodePoints(points []int) string { + scalars := make([]rune, 0, len(points)) + for _, point := range points { + scalars = append(scalars, rune(point)) + } + return string(scalars) +} + +func asAscii(value string) *string { + for _, scalar := range value { + if scalar >= 0x80 { + return nil + } + } + return &value +} + +func asDigits(value string) *string { + if value == "" { + return nil + } + for _, scalar := range value { + if scalar < '0' || scalar > '9' { + return nil + } + } + return &value +} + +func parseDigits(value string) *int { + if value == "" || len(value) > 18 { + return nil + } + total := 0 + for _, scalar := range value { + if scalar < '0' || scalar > '9' { + return nil + } + total = total*10 + int(scalar-'0') + } + return &total +} + +func padStart(value string, length int, pad string) string { + scalars := []rune(value) + if len(scalars) >= length { + return value + } + prefix := make([]rune, 0, length-len(scalars)) + for index := 0; index < length-len(scalars); index++ { + prefix = append(prefix, []rune(pad)...) + } + return string(prefix) + value +} + +func codeAt(value string, index int) *int { + if index < 0 || index >= len(value) { + return nil + } + point := int(value[index]) + return &point +} + +func charAt(value string, index int) *string { + if index < 0 || index >= len(value) { + return nil + } + scalar := string(value[index]) + return &scalar +} + +func compareInts(left int, right int) int { + if left < right { + return -1 + } + if left > right { + return 1 + } + return 0 +} + +func absInt(value int) int { + if value < 0 { + return -value + } + return value +} + +func at[T any](values []T, index int) *T { + if index < 0 || index >= len(values) { + return nil + } + return &values[index] +} + +func sumInts(values []int) int { + total := 0 + for _, value := range values { + total += value + } + return total +} + +func contains[T comparable](values []T, needle T) bool { + return slices.Contains(values, needle) +} + +func indexOf[T comparable](values []T, needle T) int { + return slices.Index(values, needle) +} + +func reversed[T any](values []T) []T { + out := make([]T, len(values)) + for index, value := range values { + out[len(values)-1-index] = value + } + return out +} + +func sortedStable[T any](values []T, compare func(T, T) int) []T { + out := make([]T, len(values)) + copy(out, values) + slices.SortStableFunc(out, compare) + return out +} + +func sortedStableBy[T any, K any](values []T, key func(T) K, compare func(K, K) int) []T { + out := make([]T, len(values)) + copy(out, values) + slices.SortStableFunc(out, func(left T, right T) int { return compare(key(left), key(right)) }) + return out +} + +func mapped[T any, R any](values []T, fn func(T) R) []R { + out := make([]R, 0, len(values)) + for _, value := range values { + out = append(out, fn(value)) + } + return out +} + +func filtered[T any](values []T, keep func(T) bool) []T { + out := make([]T, 0, len(values)) + for _, value := range values { + if keep(value) { + out = append(out, value) + } + } + return out +} + +func anyOf[T any](values []T, test func(T) bool) bool { + for _, value := range values { + if test(value) { + return true + } + } + return false +} + +func allOf[T any](values []T, test func(T) bool) bool { + for _, value := range values { + if !test(value) { + return false + } + } + return true +} + +func found[T any](values []T, test func(T) bool) *T { + for _, value := range values { + if test(value) { + return ptr(value) + } + } + return nil +} + +func dateFromEpochDays(days int) *int { + if days < -719162 || days > 2932896 { + return nil + } + return &days +} + +func dateAddDays(days int, shift int) *int { + return dateFromEpochDays(days + shift) +} + +// raceFirstSome runs idempotent tasks concurrently and takes the first one that answers. +// Cancellation is best effort: a losing goroutine may finish, and its answer is dropped. +func raceFirstSome[T any](tasks []func() *T) *T { + results := make(chan *T, len(tasks)) + for _, task := range tasks { + go func(run func() *T) { + results <- run() + }(task) + } + + for index := 0; index < len(tasks); index++ { + if answer := <-results; answer != nil { + return answer + } + } + + return nil +} + +// HttpHeader is one request or response header. +type HttpHeader struct { + Name string + Value string +} + +// HttpRequest is a request handed to the environment. +type HttpRequest struct { + Method string + Url string + Headers []HttpHeader + Body string + TimeoutMillis int +} + +// HttpResponse is a response from the environment. +type HttpResponse struct { + Status int + Headers []HttpHeader + Body string +} + +// Capabilities is everything the core needs from the outside world. +type Capabilities interface { + // A transport error or a timeout answers nil; a 4xx or 5xx status is a value. + Request(request HttpRequest) *HttpResponse + Now() int + Sleep(milliseconds int) + NextU32() int +} diff --git a/core/out/python/API.json b/core/out/python/API.json new file mode 100644 index 000000000..a72825f10 --- /dev/null +++ b/core/out/python/API.json @@ -0,0 +1,233 @@ +{ + "functions": [ + { + "name": "format_cnpj", + "module": "format_cnpj.py", + "params": [ + { + "name": "value", + "type": "str" + }, + { + "name": "options", + "type": "FormatCnpjOptions" + } + ], + "returns": "str", + "effects": [] + }, + { + "name": "format_currency", + "module": "format_currency.py", + "params": [ + { + "name": "value", + "type": "int" + }, + { + "name": "symbol", + "type": "bool" + } + ], + "returns": "str", + "effects": [] + }, + { + "name": "generate_cnpj", + "module": "generate_cnpj.py", + "params": [], + "returns": "str", + "effects": [] + }, + { + "name": "generate_cpf", + "module": "generate_cpf.py", + "params": [], + "returns": "str", + "effects": [] + }, + { + "name": "get_address_info_by_cep", + "module": "get_address_info_by_cep.py", + "params": [ + { + "name": "cep", + "type": "str" + } + ], + "returns": "AddressInfo", + "effects": [ + "Fail", + "Fail" + ] + }, + { + "name": "get_holidays", + "module": "get_holidays.py", + "params": [ + { + "name": "year", + "type": "int" + } + ], + "returns": "List[Holiday]", + "effects": [] + }, + { + "name": "is_business_day", + "module": "is_business_day.py", + "params": [ + { + "name": "value", + "type": "int" + }, + { + "name": "include_optional", + "type": "bool" + } + ], + "returns": "bool", + "effects": [] + }, + { + "name": "is_valid_cnpj", + "module": "is_valid_cnpj.py", + "params": [ + { + "name": "cnpj", + "type": "str" + }, + { + "name": "version", + "type": "Literal[\"1\", \"2\"]" + } + ], + "returns": "bool", + "effects": [] + }, + { + "name": "is_valid_cpf", + "module": "is_valid_cpf.py", + "params": [ + { + "name": "cpf", + "type": "str" + } + ], + "returns": "bool", + "effects": [] + } + ], + "seams": [ + { + "name": "generate_cnpj_with", + "publicName": "generate_cnpj", + "module": "generate_cnpj.py", + "params": [ + { + "name": "env", + "type": "Capabilities" + } + ], + "returns": "str", + "hasWrapper": true + }, + { + "name": "generate_cpf_with", + "publicName": "generate_cpf", + "module": "generate_cpf.py", + "params": [ + { + "name": "env", + "type": "Capabilities" + } + ], + "returns": "str", + "hasWrapper": true + }, + { + "name": "get_address_info_by_cep_with", + "publicName": "get_address_info_by_cep", + "module": "get_address_info_by_cep.py", + "params": [ + { + "name": "cep", + "type": "str" + }, + { + "name": "env", + "type": "Capabilities" + } + ], + "returns": "AddressInfo", + "hasWrapper": true + } + ], + "records": [ + { + "name": "FormatCnpjOptions", + "fields": [ + { + "name": "pad", + "type": "bool" + }, + { + "name": "version", + "type": "Literal[\"1\", \"2\"]" + }, + { + "name": "obfuscate", + "type": "bool" + } + ] + }, + { + "name": "AddressInfo", + "fields": [ + { + "name": "cep", + "type": "str" + }, + { + "name": "state", + "type": "str" + }, + { + "name": "city", + "type": "str" + }, + { + "name": "neighborhood", + "type": "str" + }, + { + "name": "street", + "type": "str" + } + ] + }, + { + "name": "Holiday", + "fields": [ + { + "name": "name", + "type": "str" + }, + { + "name": "date", + "type": "int" + }, + { + "name": "type", + "type": "Literal[\"national\", \"optional\", \"religious\", \"state\"]" + } + ] + } + ], + "errors": [ + "GetAddressInfoByCepError", + "GetAddressInfoByCepNotFoundError", + "GetAddressInfoByCepValidationError", + "HttpError" + ] +} diff --git a/core/out/python/LOWERING.md b/core/out/python/LOWERING.md new file mode 100644 index 000000000..cd3e214ae --- /dev/null +++ b/core/out/python/LOWERING.md @@ -0,0 +1,337 @@ +# Lowering selections — python + +Generated by the engine. Each row is one operation, the argument types it was called +with, the implementation that was selected, and the rule that decided it. + +| operation | argument types | implementation | why | +| --- | --- | --- | --- | +| `clock.sleep` | `Duration` | native | only candidate, cost none/constant | +| `core.eq` | `"1" \| "2", "2"` | native | only candidate, cost none/constant | +| `core.eq` | `"national" \| "optional" \| "religious" \| "state", "optional"` | native | only candidate, cost none/constant | +| `core.eq` | `Ascii[1], Ascii[1]` | native | only candidate, cost none/constant | +| `core.eq` | `Ascii[1], Digits[1]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[-1..1], Int[0..0]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[-2..2], Int[0..0]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[-48..79], Int[0..9]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[-1..-1]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[0..127]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[10..10]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[110..110]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[114..114]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[116..116]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[117..117]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[13..13]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[32..32]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[34..34]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[58..58]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[9..9]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[92..92]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..2147483647], Int[11..11]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..2147483647], Int[14..14]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..3506328], Int[306..3652364]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..9], Int[0..9]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..9600], Int[0..9999]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..12], Int[1..12]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..31], Int[1..31]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..7], Int[6..6]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..7], Int[7..7]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..9999], Int[1..9999]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[48..57], Int[48..57]` | native | only candidate, cost none/constant | +| `core.eq` | `String[0..2147483647], Ascii[0]` | native | only candidate, cost none/constant | +| `date.addDays` | `CivilDate, Int[-2..-2]` | native | only candidate, cost none/constant | +| `date.addDays` | `CivilDate, Int[-47..-47]` | native | only candidate, cost none/constant | +| `date.addDays` | `CivilDate, Int[60..60]` | native | only candidate, cost none/constant | +| `date.clampEpochDays` | `Int[0..0]` | native | only candidate, cost none/constant | +| `date.compare` | `CivilDate, CivilDate` | native | only candidate, cost none/constant | +| `date.dayOfWeek` | `CivilDate` | native | only candidate, cost none/constant | +| `date.fromYmd` | `Int[1900..2099], Int[1..1], Int[1..1]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[10..10], Int[12..12]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[11..11], Int[15..15]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[11..11], Int[2..2]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[12..12], Int[25..25]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[3..3], Int[22..31]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[4..4], Int[1..25]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[4..4], Int[21..21]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[5..5], Int[1..1]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[9..9], Int[7..7]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[2024..2099], Int[11..11], Int[20..20]` | portable | only candidate, cost none/linear | +| `date.year` | `CivilDate` | portable | only candidate, cost none/constant | +| `dec.abs` | `Decimal<2>` | native | only candidate, cost none/constant | +| `dec.isNegative` | `Decimal<2>` | native | only candidate, cost none/constant | +| `dec.unscaled` | `Decimal<2>` | native | only candidate, cost none/constant | +| `http.request` | `HttpRequest` | native | only candidate, cost many/linear | +| `int.add` | `Int[-10012..20013], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-146097..3506328], Int[-3506503..3798697]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-14618796..14618800], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-238871..9], Int[3..3]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-3504000..3795635], Int[-2400..2599]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-3506503..3798330], Int[0..367]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-3508380..3800745], Int[-2403..2603]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-3508623..3800862], Int[-95..103]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-36547263..36546651], Int[2..2]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-36547330..36546740], Int[2..2]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-7..35], Int[114..114]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-719162..2932896], Int[719468..719468]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-9612..10413], Int[-400..9600]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..1683], Int[2..2]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..17], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..18], Int[0..319]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..2147483646], Int[0..4]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..2147483646], Int[5..5]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..29], Int[0..6]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..30], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..337], Int[0..132]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..337], Int[1..31]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..342], Int[19..20]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..65520], Int[0..15]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..720], Int[0..90]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..891], Int[0..81]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..891], Int[0..99]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..9007199254740991], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..9007199254740991], Int[2..2]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..9007199254740991], Int[6..6]` | native | only candidate, cost none/constant | +| `int.add` | `Int[1..12], Int[9..9]` | native | only candidate, cost none/constant | +| `int.add` | `Int[1..31], Int[0..31]` | native | only candidate, cost none/constant | +| `int.add` | `Int[32..32], Int[0..6]` | native | only candidate, cost none/constant | +| `int.add` | `Int[32..38], Int[0..48]` | native | only candidate, cost none/constant | +| `int.add` | `Int[5..2147483658], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[5..2147483659], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[6..2147483667], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[6..2147483668], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[8..352], Int[15..15]` | native | only candidate, cost none/constant | +| `int.add` | `Int[9..2147483671], Int[0..3]` | native | only candidate, cost none/constant | +| `int.div` | `Int[-3506022..3798461], Int[1460..1460]` | library | library, cost none/constant; rejected `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[-3506022..3798461], Int[146096..146096]` | library | library, cost none/constant; rejected `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[-3506022..3798461], Int[36524..36524]` | library | library, cost none/constant; rejected `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[-3508743..3800988], Int[365..365]` | library | library, cost none/constant; rejected `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[-36547261..36546653], Int[5..5]` | library | library, cost none/constant; rejected `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[-36547328..36546742], Int[153..153]` | library | library, cost none/constant; rejected `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[-9600..10399], Int[100..100]` | library | library, cost none/constant; rejected `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[-9600..10399], Int[4..4]` | library | library, cost none/constant; rejected `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[-9612..10413], Int[100..100]` | library | library, cost none/constant; rejected `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[-9612..10413], Int[4..4]` | library | library, cost none/constant; rejected `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[0..469], Int[451..451]` | native | only candidate, cost none/constant; `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[0..99], Int[4..4]` | native | only candidate, cost none/constant; `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[0..9999], Int[400..400]` | native | only candidate, cost none/constant; `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[107..149], Int[31..31]` | native | only candidate, cost none/constant; `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[19..20], Int[4..4]` | native | only candidate, cost none/constant; `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[1900..2099], Int[100..100]` | native | only candidate, cost none/constant; `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[2..1685], Int[5..5]` | native | only candidate, cost none/constant; `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.div` | `Int[306..3652364], Int[146097..146097]` | native | only candidate, cost none/constant; `//` floors, so it only matches truncated division when both operands are non-negative | +| `int.ge` | `Int[0..1114111], Int[0..0]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..1114111], Int[48..48]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..1114111], Int[65..65]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..1114111], Int[97..97]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..127], Int[65..65]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..17], Int[0..2147483647]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..599], Int[200..200]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[1900..2099], Int[2024..2024]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[0..14], Int[0..0]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[0..2], Int[0..0]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1..12], Int[2..2]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1..9007199254740991], Int[12..12]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1..9007199254740991], Int[31..31]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1..9007199254740991], Int[9999..9999]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1900..9999], Int[2099..2099]` | native | only candidate, cost none/constant | +| `int.le` | `Int[-238868..238858], Int[2..2]` | native | only candidate, cost none/constant | +| `int.le` | `Int[1..12], Int[2..2]` | native | only candidate, cost none/constant | +| `int.le` | `Int[22..56], Int[31..31]` | native | only candidate, cost none/constant | +| `int.le` | `Int[48..1114111], Int[57..57]` | native | only candidate, cost none/constant | +| `int.le` | `Int[65..1114111], Int[70..70]` | native | only candidate, cost none/constant | +| `int.le` | `Int[65..127], Int[90..90]` | native | only candidate, cost none/constant | +| `int.le` | `Int[97..1114111], Int[102..102]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[-238871..238867], Int[10..10]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[-9007199254740991..9007199254740991], Int[1..1]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[0..10], Int[2..2]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[0..17], Int[0..2147483647]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[0..4294967295], Int[4294967287..4294967296]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[0..9999], Int[0..0]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[1..9999], Int[1900..1900]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[200..599], Int[300..300]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[306..3652364], Int[0..0]` | native | only candidate, cost none/constant | +| `int.max` | `Int[-10012..20014], Int[1..1]` | native | only candidate, cost none/constant | +| `int.max` | `Int[-14618795..14618801], Int[1..1]` | native | only candidate, cost none/constant | +| `int.max` | `Int[-238868..238858], Int[1..1]` | native | only candidate, cost none/constant | +| `int.max` | `Int[-4372068..6585557], Int[-719162..-719162]` | native | only candidate, cost none/constant | +| `int.max` | `Int[1..15], Int[0..0]` | native | only candidate, cost none/constant | +| `int.max` | `Int[1..62], Int[22..22]` | native | only candidate, cost none/constant | +| `int.min` | `Int[-719162..6585557], Int[2932896..2932896]` | native | only candidate, cost none/constant | +| `int.min` | `Int[0..65535], Int[65535..65535]` | native | only candidate, cost none/constant | +| `int.min` | `Int[1..14618801], Int[31..31]` | native | only candidate, cost none/constant | +| `int.min` | `Int[1..20014], Int[9999..9999]` | native | only candidate, cost none/constant | +| `int.min` | `Int[1..238858], Int[12..12]` | native | only candidate, cost none/constant | +| `int.min` | `Int[22..62], Int[56..56]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[-14..14], Int[3..3]` | library | library, cost none/constant; rejected `%` is floored in Python, so it only matches the Core's truncated remainder for non-negative operands | +| `int.mod` | `Int[0..4294967295], Int[10..10]` | native | only candidate, cost none/constant; `%` is floored in Python, so it only matches the Core's truncated remainder for non-negative operands | +| `int.mod` | `Int[0..810], Int[11..11]` | native | only candidate, cost none/constant; `%` is floored in Python, so it only matches the Core's truncated remainder for non-negative operands | +| `int.mod` | `Int[0..86], Int[7..7]` | native | only candidate, cost none/constant; `%` is floored in Python, so it only matches the Core's truncated remainder for non-negative operands | +| `int.mod` | `Int[0..972], Int[11..11]` | native | only candidate, cost none/constant; `%` is floored in Python, so it only matches the Core's truncated remainder for non-negative operands | +| `int.mod` | `Int[0..99], Int[4..4]` | native | only candidate, cost none/constant; `%` is floored in Python, so it only matches the Core's truncated remainder for non-negative operands | +| `int.mod` | `Int[0..990], Int[11..11]` | native | only candidate, cost none/constant; `%` is floored in Python, so it only matches the Core's truncated remainder for non-negative operands | +| `int.mod` | `Int[107..149], Int[31..31]` | native | only candidate, cost none/constant; `%` is floored in Python, so it only matches the Core's truncated remainder for non-negative operands | +| `int.mod` | `Int[19..20], Int[4..4]` | native | only candidate, cost none/constant; `%` is floored in Python, so it only matches the Core's truncated remainder for non-negative operands | +| `int.mod` | `Int[1900..2099], Int[100..100]` | native | only candidate, cost none/constant; `%` is floored in Python, so it only matches the Core's truncated remainder for non-negative operands | +| `int.mod` | `Int[1900..2099], Int[19..19]` | native | only candidate, cost none/constant; `%` is floored in Python, so it only matches the Core's truncated remainder for non-negative operands | +| `int.mod` | `Int[23..367], Int[30..30]` | native | only candidate, cost none/constant; `%` is floored in Python, so it only matches the Core's truncated remainder for non-negative operands | +| `int.mul` | `Int[-1..24], Int[146097..146097]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[-1..24], Int[400..400]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[-9600..10399], Int[365..365]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..1], Int[31..31]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..24], Int[146097..146097]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..24], Int[400..400]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..4095], Int[16..16]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..9], Int[2..10]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..9], Int[2..11]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..9], Int[2..9]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[11..11], Int[0..29]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[153..153], Int[-238871..238867]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[153..153], Int[0..11]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[19..19], Int[0..18]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[2..2], Int[0..24]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[2..2], Int[0..3]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[22..22], Int[0..6]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[365..365], Int[-9612..10413]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[5..5], Int[-7309466..7309348]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[7..7], Int[0..1]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3506022..3798461], Int[-2401..2601]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3506022..3798461], Int[-3510887..3803444]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3506400..3798234], Int[-96..103]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3508718..3800965], Int[-23..25]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3510783..3803348], Int[-96..104]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3652600..7305025], Int[719468..719468]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-7309466..7309348], Int[-7309452..7309330]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..127], Int[48..48]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..15], Int[1..14]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..24], Int[1..1]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..35], Int[0..7]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..9999], Int[-400..9600]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[1..12], Int[3..3]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[1..368], Int[1..1]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[1..9999], Int[1..1]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[10..10], Int[0..8]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[10..238867], Int[9..9]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[11..11], Int[0..9]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[11..11], Int[2..10]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[14..358], Int[6..6]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[19..362], Int[4..5]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[3..17], Int[2..2]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[3..4], Int[3..3]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[3..86], Int[0..3]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[306..3652364], Int[-146097..3506328]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[32..56], Int[31..31]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[32..86], Int[0..29]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[48..57], Int[48..48]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[65..70], Int[55..55]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[97..102], Int[87..87]` | native | only candidate, cost none/constant | +| `opt.isNone` | `Option` | native | only candidate, cost none/constant | +| `opt.isNone` | `Option` | native | only candidate, cost none/constant | +| `opt.isNone` | `Option` | native | only candidate, cost none/constant | +| `opt.isNone` | `Option` | native | only candidate, cost none/constant | +| `opt.isNone` | `Option` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Ascii[0]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, CivilDate` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[-1..-1]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[0..0]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[48..48]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[-2..-2]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[0..0]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[48..48]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Ascii[0]` | native | only candidate, cost none/constant | +| `opt.unwrap` | `Option` | native | only candidate, cost none/constant | +| `opt.unwrap` | `Option` | native | only candidate, cost none/constant | +| `opt.unwrap` | `Option` | native | only candidate, cost none/constant | +| `opt.unwrap` | `Option` | native | only candidate, cost none/constant | +| `opt.unwrap` | `Option` | native | only candidate, cost none/constant | +| `random.nextU32` | `` | native | only candidate, cost none/constant | +| `re.retain` | `Ascii[1..15]` | native | only candidate, cost one/linear; re.sub with the negated class is one pass | +| `re.retain` | `String[0..2147483647]` | native | only candidate, cost one/linear; re.sub with the negated class is one pass | +| `re.test` | `String[0..2147483647]` | native | only candidate, cost none/linear; `fullmatch` anchors the whole string, and the normalized pattern uses explicit classes | +| `seq.at` | `List[0..2147483647], Int[0..2147483650]` | native | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[0..9007199254740991]` | native | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[1..9007199254740992]` | native | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[5..2147483658]` | native | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[5..2147483659]` | native | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[6..2147483667]` | native | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[6..2147483668]` | native | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[9..2147483674]` | native | only candidate, cost none/constant | +| `seq.at` | `List[5..5], Int[0..4]` | native | only candidate, cost none/constant | +| `seq.at` | `List[0..15], Int[0..14]` | native | only candidate, cost none/constant | +| `seq.get` | `List[12..12], Int[0..11]` | native | only candidate, cost none/constant | +| `seq.get` | `List[13..13], Int[0..11]` | native | only candidate, cost none/constant | +| `seq.len` | `List[0..2147483647]` | native | only candidate, cost none/constant | +| `seq.len` | `List[5..5]` | native | only candidate, cost none/constant | +| `seq.len` | `List[12..12]` | native | only candidate, cost none/constant | +| `seq.len` | `List[0..15]` | native | only candidate, cost none/constant | +| `seq.push` | `` | native | only candidate, cost none/constant | +| `seq.sortStableBy` | `List[12..13], (Holiday) => CivilDate` | native | only candidate, cost one/nlogn; `sorted(key=…)` is stable and avoids the comparator wrapper | +| `str.asciiUpper` | `Ascii[0..2147483647]` | native | only candidate, cost one/linear; `upper()` is only ASCII-equivalent on ASCII input | +| `str.asciiUpper` | `String[0..2147483647]` | native | native, cost one/linear; one re.sub pass maps a-z and leaves every other scalar alone; rejected `upper()` is only ASCII-equivalent on ASCII input | +| `str.charAtOpt` | `Ascii[0..2147483647], Int[0..17]` | native | only candidate, cost none/constant | +| `str.charAtOpt` | `Ascii[18], Int[0..17]` | native | only candidate, cost none/constant | +| `str.codeAt` | `Ascii[14], Int[12..12]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAt` | `Ascii[14], Int[13..13]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAt` | `Digits[11], Int[0..0]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAt` | `Digits[11], Int[0..8]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAt` | `Digits[11], Int[0..9]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAt` | `Digits[11], Int[1..10]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAt` | `Digits[11], Int[10..10]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAt` | `Digits[11], Int[9..9]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAt` | `Digits[12], Int[0..0]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAt` | `Digits[12], Int[1..11]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAt` | `Digits[14], Int[0..0]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAt` | `Digits[14], Int[0..11]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAt` | `Digits[14], Int[1..13]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAt` | `Digits[14], Int[12..12]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAt` | `Digits[14], Int[13..13]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAt` | `Digits[9], Int[0..0]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAt` | `Digits[9], Int[1..11]` | native | only candidate, cost none/constant; indexing by scalar is O(1) and identical to the Core only for ASCII | +| `str.codeAtOpt` | `Ascii[0..2147483647], Int[0..2147483646]` | native | only candidate, cost none/constant | +| `str.codePoints` | `Ascii[5]` | native | only candidate, cost one/linear | +| `str.codePoints` | `Digits[0..15]` | native | only candidate, cost one/linear | +| `str.codePoints` | `String[0..2147483647]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[0..17], Ascii[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[0..3], Ascii[1..47]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[0..30], Ascii[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[1..31], Ascii[0..16]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[1..4], Ascii[1..47]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[1], Ascii[0..3]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[1], Ascii[3]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[2], Ascii[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[25], Digits[8] matches ^[0-9]{8}$` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[33], Ascii[6]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[36], Digits[8] matches ^[0-9]{8}$` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[4], Ascii[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[1], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[10], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[11], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[12], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[12], Digits[2]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[13], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[2], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[3], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[4], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[5], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[6], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[7], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[8], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[9], Digits[1]` | native | only candidate, cost one/linear | +| `str.concat` | `Digits[9], Digits[2]` | native | only candidate, cost one/linear | +| `str.fromCodePoints` | `List[0..2147483647]` | native | only candidate, cost one/linear | +| `str.fromCodePoints` | `List[0..30]` | native | only candidate, cost one/linear | +| `str.fromCodePoints` | `List[1..1]` | native | only candidate, cost one/linear | +| `str.fromInt` | `Int[-9007199254740991..9007199254740991]` | native | only candidate, cost one/linear | +| `str.fromInt` | `Int[0..9]` | native | only candidate, cost one/linear | +| `str.len` | `Ascii[0..2147483647]` | native | only candidate, cost none/constant; `len` counts code points in Python, which is the Core's definition | +| `str.len` | `Ascii[18]` | native | only candidate, cost none/constant; `len` counts code points in Python, which is the Core's definition | +| `str.len` | `Ascii[3..17]` | native | only candidate, cost none/constant; `len` counts code points in Python, which is the Core's definition | +| `str.len` | `Digits[0..2147483647]` | native | only candidate, cost none/constant; `len` counts code points in Python, which is the Core's definition | +| `str.len` | `Digits[12]` | native | only candidate, cost none/constant; `len` counts code points in Python, which is the Core's definition | +| `str.len` | `Digits[9]` | native | only candidate, cost none/constant; `len` counts code points in Python, which is the Core's definition | +| `str.padStart` | `Ascii[0..2147483647], Int[0..18], Digits[1]` | native | only candidate, cost one/linear | +| `str.padStart` | `Ascii[1..17], Int[3..3], Digits[1]` | native | only candidate, cost one/linear | +| `str.slice` | `Ascii[3..17], Int[0..0], Int[1..15]` | native | only candidate, cost one/linear | +| `str.slice` | `Ascii[3..17], Int[1..15], Int[3..17]` | native | only candidate, cost one/linear | +| `str.trim` | `String[0..2147483647]` | native | only candidate, cost one/linear; `strip()` uses Python's own whitespace set, so the 25 code points are passed explicitly | +| `task.race` | `List<() => Option>[2..2]` | library | only candidate, cost many/linear; a ThreadPoolExecutor is the standard library's way to run idempotent requests concurrently | + +Mix: 304 native, 12 library, 12 portable. diff --git a/core/out/python/SOURCEMAP.json b/core/out/python/SOURCEMAP.json new file mode 100644 index 000000000..2975e175f --- /dev/null +++ b/core/out/python/SOURCEMAP.json @@ -0,0 +1,137 @@ +{ + "lib/format.py#pattern_slots": { + "module": "lib/format", + "start": 1345, + "end": 2086 + }, + "lib/format.py#format_with_pattern": { + "module": "lib/format", + "start": 2182, + "end": 2821 + }, + "format_cnpj.py#format_cnpj": { + "module": "format-cnpj", + "start": 967, + "end": 1233 + }, + "format_currency.py#format_currency": { + "module": "format-currency", + "start": 966, + "end": 1499 + }, + "generate_cnpj.py#generate_cnpj": { + "module": "generate-cnpj", + "start": 982, + "end": 1571 + }, + "generate_cnpj.py#generate_cnpj_with": { + "module": "generate-cnpj", + "start": 982, + "end": 1571 + }, + "generate_cpf.py#generate_cpf": { + "module": "generate-cpf", + "start": 1014, + "end": 1566 + }, + "generate_cpf.py#generate_cpf_with": { + "module": "generate-cpf", + "start": 1014, + "end": 1566 + }, + "get_address_info_by_cep.py#get_with_retry": { + "module": "get-address-info-by-cep", + "start": 1086, + "end": 1491 + }, + "get_address_info_by_cep.py#is_ok": { + "module": "get-address-info-by-cep", + "start": 1529, + "end": 1622 + }, + "get_address_info_by_cep.py#fetch_via_cep": { + "module": "get-address-info-by-cep", + "start": 1701, + "end": 2300 + }, + "get_address_info_by_cep.py#fetch_brasil_api": { + "module": "get-address-info-by-cep", + "start": 2351, + "end": 2957 + }, + "get_address_info_by_cep.py#get_address_info_by_cep": { + "module": "get-address-info-by-cep", + "start": 3382, + "end": 3789 + }, + "get_address_info_by_cep.py#get_address_info_by_cep_with": { + "module": "get-address-info-by-cep", + "start": 3382, + "end": 3789 + }, + "lib/json.py#json_string_field": { + "module": "lib/json", + "start": 2300, + "end": 4239 + }, + "lib/civil.py#civil_date_10": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "get_holidays.py#get_holidays": { + "module": "get-holidays", + "start": 898, + "end": 2426 + }, + "is_business_day.py#is_business_day": { + "module": "is-business-day", + "start": 578, + "end": 1064 + }, + "lib/cnpj.py#cnpj_check_digit": { + "module": "lib/cnpj", + "start": 687, + "end": 1184 + }, + "lib/cnpj.py#is_repeated_cnpj": { + "module": "lib/cnpj", + "start": 3028, + "end": 3247 + }, + "is_valid_cnpj.py#is_valid_cnpj": { + "module": "is-valid-cnpj", + "start": 1397, + "end": 2445 + }, + "lib/cpf.py#cpf_check_digit_1": { + "module": "lib/cpf", + "start": 394, + "end": 687 + }, + "is_valid_cpf.py#is_valid_cpf": { + "module": "is-valid-cpf", + "start": 793, + "end": 1143 + }, + "std/date.py#month_from_days": { + "module": "std/date", + "start": 2280, + "end": 2780 + }, + "std/date.py#day_from_days": { + "module": "std/date", + "start": 2855, + "end": 3346 + }, + "std/date.py#ymd_to_days": { + "module": "std/date", + "start": 3922, + "end": 4284 + }, + "std/date.py#year_from_days": { + "module": "std/date", + "start": 1627, + "end": 2212 + } +} diff --git a/core/out/python/__init__.py b/core/out/python/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/core/out/python/_driver.py b/core/out/python/_driver.py new file mode 100644 index 000000000..3581d44c2 --- /dev/null +++ b/core/out/python/_driver.py @@ -0,0 +1,133 @@ +# Code generated by the logic engine. DO NOT EDIT. +# source: _driver + +import dataclasses +import json +import sys + + +def encode(value: object) -> object: + """A generated record is a dataclass, which json.dumps does not know.""" + if dataclasses.is_dataclass(value): + return dataclasses.asdict(value) + raise TypeError(value) + + +from .format_cnpj import FormatCnpjOptions, format_cnpj +from .format_currency import format_currency +from .generate_cnpj import generate_cnpj_with +from .generate_cpf import generate_cpf_with +from .get_address_info_by_cep import get_address_info_by_cep_with +from .get_holidays import get_holidays +from .is_business_day import is_business_day +from .is_valid_cnpj import is_valid_cnpj +from .is_valid_cpf import is_valid_cpf +import os +import time +from typing import Optional +from ._support import Capabilities, HttpRequest, HttpResponse + + +# The reference PCG32: same constants and default seed as the interpreter's, so a draw +# matches the reference bit for bit. A fresh instance is built for every request, the same +# way the reference model starts a fresh interpreter, and so a fresh generator, per case. +class Pcg32: + MASK64 = (1 << 64) - 1 + MASK32 = (1 << 32) - 1 + INCREMENT = 1442695040888963407 + + def __init__(self, seed: int) -> None: + self.state = 0 + self.next_u32() + self.state = (self.state + seed) & self.MASK64 + self.next_u32() + + def next_u32(self) -> int: + previous = self.state + self.state = (previous * 6364136223846793005 + self.INCREMENT) & self.MASK64 + xorshifted = (((previous >> 18) ^ previous) >> 27) & self.MASK32 + rotation = previous >> 59 + return ( + (xorshifted >> rotation) | (xorshifted << ((-rotation) & 31)) + ) & self.MASK32 + + +# The interpreter's own default: its constructor falls back to this seed whenever +# `Capabilities.seed` is left unset, which is how every conformance case runs it. +DEFAULT_SEED = 0x853C49E6748FEA9B + + +class FakeCapabilities: + """The capability fake the differential harness drives: responses come from fixtures.json.""" + + def __init__(self, fixtures: dict) -> None: + self.fixtures = fixtures + self.random = Pcg32(DEFAULT_SEED) + + def request(self, request: HttpRequest) -> Optional[HttpResponse]: + fixture = self.fixtures.get(request.url) + + if fixture is None: + return None + + time.sleep(fixture.get("latencyMillis", 0) / 1000) + + return HttpResponse(status=fixture["status"], headers=[], body=fixture["body"]) + + def now(self) -> int: + return 0 + + def sleep(self, milliseconds: int) -> None: + time.sleep(milliseconds / 1000) + + def next_u32(self) -> int: + return self.random.next_u32() + + +FIXTURE_PATH = os.path.join(os.path.dirname(__file__), "fixtures.json") + +if os.path.exists(FIXTURE_PATH): + with open(FIXTURE_PATH) as handle: + FIXTURES = json.load(handle) +else: + FIXTURES = None + +HANDLERS = { + "format-cnpj::formatCnpj": lambda args: format_cnpj( + args[0], + FormatCnpjOptions( + pad=args[1]["pad"], + version=args[1]["version"], + obfuscate=args[1]["obfuscate"], + ), + ), + "format-currency::formatCurrency": lambda args: format_currency(args[0], args[1]), + "generate-cnpj::generateCnpj": lambda args: generate_cnpj_with(ENVIRONMENT), + "generate-cpf::generateCpf": lambda args: generate_cpf_with(ENVIRONMENT), + "get-address-info-by-cep::getAddressInfoByCep": lambda args: ( + get_address_info_by_cep_with(args[0], ENVIRONMENT) + ), + "get-holidays::getHolidays": lambda args: get_holidays(args[0]), + "is-business-day::isBusinessDay": lambda args: is_business_day(args[0], args[1]), + "is-valid-cnpj::isValidCnpj": lambda args: is_valid_cnpj(args[0], args[1]), + "is-valid-cpf::isValidCpf": lambda args: is_valid_cpf(args[0]), +} + + +for line in sys.stdin: + if line.strip() == "": + continue + + request = json.loads(line) + + # A fresh environment per line: next_u32 starts from the same state the reference + # model's fresh interpreter starts from for every case. + ENVIRONMENT = FakeCapabilities(FIXTURES) if FIXTURES is not None else Capabilities() + + try: + value = HANDLERS[request["fn"]](request["args"]) + print(json.dumps({"ok": True, "value": value}, default=encode)) + except Exception as error: + print(json.dumps({"ok": False, "error": type(error).__name__})) + + sys.stdout.flush() diff --git a/core/out/python/_support.py b/core/out/python/_support.py new file mode 100644 index 000000000..51600ab5f --- /dev/null +++ b/core/out/python/_support.py @@ -0,0 +1,101 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: _support + +from dataclasses import dataclass +from typing import Callable, List, Optional, Sequence, TypeVar + +T = TypeVar("T") + + +def race_first_some(tasks: Sequence[Callable[[], Optional[T]]]) -> Optional[T]: + from concurrent.futures import ThreadPoolExecutor, as_completed + + with ThreadPoolExecutor(max_workers=max(1, len(tasks))) as pool: + futures = [pool.submit(task) for task in tasks] + + for future in as_completed(futures): + try: + return future.result() + except Exception: + continue + + return None + + +@dataclass(frozen=True) +class HttpHeader: + name: str + value: str + + +@dataclass(frozen=True) +class HttpRequest: + method: str + url: str + headers: List[HttpHeader] + body: str + timeout_millis: int + + +@dataclass(frozen=True) +class HttpResponse: + status: int + headers: List[HttpHeader] + body: str + + +class Capabilities: + """The default environment, built from the standard library only.""" + + def request(self, request: HttpRequest) -> Optional[HttpResponse]: + import json + import urllib.error + import urllib.request + + payload = None if request.body == "" else request.body.encode() + parsed = urllib.request.Request( + request.url, data=payload, method=request.method + ) + + for header in request.headers: + parsed.add_header(header.name, header.value) + + try: + with urllib.request.urlopen( + parsed, timeout=request.timeout_millis / 1000 + ) as response: + return HttpResponse( + status=response.status, headers=[], body=response.read().decode() + ) + except urllib.error.HTTPError as error: + return HttpResponse( + status=error.code, headers=[], body=error.read().decode() + ) + except Exception: + return None + + def now(self) -> int: + import time + + return int(time.time() * 1000) + + def sleep(self, milliseconds: int) -> None: + import time + + time.sleep(milliseconds / 1000) + + def next_u32(self) -> int: + import random + + # Not cryptographically secure, deliberately: the utilities that draw are + # generating example documents, and that is what the published package + # documents doing. A caller who needs unpredictability passes its own + # capability, the way the conformance harness passes a seeded one. + return random.getrandbits(32) + + +# The platform default, built once at import time rather than per call — every public +# wrapper (docs/decisions/0011-public-entry-points-vs-capabilities.md) shares this one +# instance, the same way a caller who builds their own environment would share it. +DEFAULT_CAPABILITIES = Capabilities() diff --git a/core/out/python/errors.py b/core/out/python/errors.py new file mode 100644 index 000000000..215f68d17 --- /dev/null +++ b/core/out/python/errors.py @@ -0,0 +1,23 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: errors + + +class DomainError(Exception): + """The root of every domain error the core raises.""" + + +class HttpError(DomainError): + """Raised by the engine's own intrinsics.""" + + +class GetAddressInfoByCepError(DomainError): + """Base of every error this utility raises.""" + + +class GetAddressInfoByCepValidationError(GetAddressInfoByCepError): + """The value given is not a CEP.""" + + +class GetAddressInfoByCepNotFoundError(GetAddressInfoByCepError): + """No CEP service knows this CEP, or none answered.""" diff --git a/core/out/python/fixtures.json b/core/out/python/fixtures.json new file mode 100644 index 000000000..f05926010 --- /dev/null +++ b/core/out/python/fixtures.json @@ -0,0 +1,27 @@ +{ + "https://viacep.com.br/ws/01310100/json/": { + "status": 200, + "body": "{\"cep\":\"01310-100\",\"logradouro\":\"Avenida Paulista\",\"bairro\":\"Bela Vista\",\"localidade\":\"São Paulo\",\"uf\":\"SP\"}", + "latencyMillis": 60 + }, + "https://brasilapi.com.br/api/cep/v1/01310100": { + "status": 200, + "body": "{\"cep\":\"01310100\",\"state\":\"SP\",\"city\":\"São Paulo\",\"neighborhood\":\"Bela Vista\",\"street\":\"Avenida Paulista\"}", + "latencyMillis": 20 + }, + "https://brasilapi.com.br/api/cep/v1/30130010": { + "status": 200, + "body": "{\"cep\":\"30130010\",\"state\":\"MG\",\"city\":\"Belo Horizonte\",\"neighborhood\":\"Centro\",\"street\":\"Avenida Afonso Pena\"}", + "latencyMillis": 40 + }, + "https://viacep.com.br/ws/99999999/json/": { + "status": 200, + "body": "{\"erro\":true}", + "latencyMillis": 10 + }, + "https://brasilapi.com.br/api/cep/v1/99999999": { + "status": 404, + "body": "{\"message\":\"not found\"}", + "latencyMillis": 10 + } +} diff --git a/core/out/python/format_cnpj.py b/core/out/python/format_cnpj.py new file mode 100644 index 000000000..47f905dd1 --- /dev/null +++ b/core/out/python/format_cnpj.py @@ -0,0 +1,41 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: format-cnpj +# content: 7541034a1f26 + +from typing import Literal +from dataclasses import dataclass +import re +from .lib.format import format_with_pattern + +__all__ = ["format_cnpj"] + + +@dataclass(frozen=True) +class FormatCnpjOptions: + pad: bool + version: Literal["1", "2"] + obfuscate: bool + + +_FORMAT_CNPJ_PATTERN_1 = re.compile("[^0-9A-Za-z]") + +_FORMAT_CNPJ_PATTERN_2 = re.compile("[^0-9]") + + +def format_cnpj(value: str, options: FormatCnpjOptions) -> str: + """Formats a CNPJ value as `00.000.000/0000-00`. + + The core takes a string and a fully normalized options record; reading a number, a missing + options object or a truthy non-boolean is the DX's job. + """ + sanitized: str = ( + _FORMAT_CNPJ_PATTERN_1.sub("", value).upper() + if (options.version == "2") + else _FORMAT_CNPJ_PATTERN_2.sub("", value) + ) + return format_with_pattern( + sanitized, + ("**.000.000/0000-**" if options.obfuscate else "00.000.000/0000-00"), + options.pad, + ) diff --git a/core/out/python/format_currency.py b/core/out/python/format_currency.py new file mode 100644 index 000000000..13bc89017 --- /dev/null +++ b/core/out/python/format_currency.py @@ -0,0 +1,57 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: format-currency +# content: 1e0947bdf328 + +from typing import List +import re + +__all__ = ["format_currency"] + +_FORMAT_CURRENCY_PATTERN_1 = re.compile("[^0-9]") + + +def format_currency(value: int, symbol: bool) -> str: + """Formats an exact amount in Brazilian Real, with two decimal places. + + The separators are the ones Lei nº 9.069/1995 art. 1º prescribes and the CLDR pt-BR data uses: + `.` between thousands, `,` before the centavos, and a non-breaking space after `R$`. A negative + amount puts the sign before the symbol, `-R$ 10,50`, the shape `Intl.NumberFormat` produces. + + Turning a host value into an exact amount is the DX's job, and so is the rounding that + conversion needs; see docs/contracts.md, which records exactly how the published package rounds. + """ + negative: bool = value < 0 + unscaled: int = abs(value) + digits: str = str(unscaled).rjust(3, "0") + cut: int = max((len(digits) - 2), 0) + whole: str = digits[0:cut] + cents: str = digits[cut : len(digits)] + __inl36_whole: str = _FORMAT_CURRENCY_PATTERN_1.sub("", whole) + __inl33_out: List[int] = [] + __inl34_scalars: List[int] = [ord(__c) for __c in __inl36_whole] + for __inl35_index in range(0, len(__inl34_scalars)): + if (__inl35_index > 0) and ( + ( + -(abs(__tm_a) % abs(__tm_b)) + if ( + (__tm_a := (len(__inl34_scalars) - __inl35_index)), + (__tm_b := 3), + __tm_a < 0, + )[2] + else abs(__tm_a) % abs(__tm_b) + ) + == 0 + ): + __inl33_out.append(46) + __inl33_out.append( + ( + __inl34_scalars[__inl35_index] + if 0 <= __inl35_index < len(__inl34_scalars) + else 48 + ) + ) + __inl37_result: str = "".join(chr(__p) for __p in __inl33_out) + body: str = (__inl37_result + ",") + cents + prefix: str = ("R$" + "".join(chr(__p) for __p in [32])) if symbol else "" + return (("-" + prefix) + body) if negative else (prefix + body) diff --git a/core/out/python/generate_cnpj.py b/core/out/python/generate_cnpj.py new file mode 100644 index 000000000..5c8223e2a --- /dev/null +++ b/core/out/python/generate_cnpj.py @@ -0,0 +1,362 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: generate-cnpj +# content: 41b441a54f18 + +from typing import Optional +from ._support import Capabilities, DEFAULT_CAPABILITIES + +__all__ = ["generate_cnpj", "generate_cnpj_with"] + + +def generate_cnpj() -> str: + """Generates a valid random CNPJ (Cadastro Nacional da Pessoa Jurídica) in the numeric format: 14 + digits, under the check digit rule both CNPJ versions share. + + Matches the published `generateCnpj()` called with no options: a random 8-digit root and + 4-digit branch (the "número de ordem"), redrawn while every digit of the 12-digit base is the + same, followed by its two check digits. The alphanumeric version and a chosen branch are DX + concerns layered on the same base and check digit rule, not a different generator. + """ + return generate_cnpj_with(DEFAULT_CAPABILITIES) + + +def generate_cnpj_with(env: Capabilities) -> str: + """`generate_cnpj`, taking its capabilities explicitly. + + The public `generate_cnpj` calls this with the platform's defaults. Pass your own to + supply a clock, a source of randomness or an HTTP client — which is what the + differential conformance driver does to make a run reproducible. + """ + __inl234_result: Optional[int] = None + __inl231_limit: int = 4294967290 + for __inl232_attempt in range(0, 32): + __inl233_draw: int = env.next_u32() + if __inl233_draw < __inl231_limit: + __inl234_result = __inl233_draw % 10 + if __inl234_result is not None: + break + if __inl234_result is None: + __inl234_result = env.next_u32() % 10 + __inl238_result: Optional[int] = None + __inl235_limit: int = 4294967290 + for __inl236_attempt in range(0, 32): + __inl237_draw: int = env.next_u32() + if __inl237_draw < __inl235_limit: + __inl238_result = __inl237_draw % 10 + if __inl238_result is not None: + break + if __inl238_result is None: + __inl238_result = env.next_u32() % 10 + __inl242_result: Optional[int] = None + __inl239_limit: int = 4294967290 + for __inl240_attempt in range(0, 32): + __inl241_draw: int = env.next_u32() + if __inl241_draw < __inl239_limit: + __inl242_result = __inl241_draw % 10 + if __inl242_result is not None: + break + if __inl242_result is None: + __inl242_result = env.next_u32() % 10 + __inl246_result: Optional[int] = None + __inl243_limit: int = 4294967290 + for __inl244_attempt in range(0, 32): + __inl245_draw: int = env.next_u32() + if __inl245_draw < __inl243_limit: + __inl246_result = __inl245_draw % 10 + if __inl246_result is not None: + break + if __inl246_result is None: + __inl246_result = env.next_u32() % 10 + __inl250_result: Optional[int] = None + __inl247_limit: int = 4294967290 + for __inl248_attempt in range(0, 32): + __inl249_draw: int = env.next_u32() + if __inl249_draw < __inl247_limit: + __inl250_result = __inl249_draw % 10 + if __inl250_result is not None: + break + if __inl250_result is None: + __inl250_result = env.next_u32() % 10 + __inl254_result: Optional[int] = None + __inl251_limit: int = 4294967290 + for __inl252_attempt in range(0, 32): + __inl253_draw: int = env.next_u32() + if __inl253_draw < __inl251_limit: + __inl254_result = __inl253_draw % 10 + if __inl254_result is not None: + break + if __inl254_result is None: + __inl254_result = env.next_u32() % 10 + __inl258_result: Optional[int] = None + __inl255_limit: int = 4294967290 + for __inl256_attempt in range(0, 32): + __inl257_draw: int = env.next_u32() + if __inl257_draw < __inl255_limit: + __inl258_result = __inl257_draw % 10 + if __inl258_result is not None: + break + if __inl258_result is None: + __inl258_result = env.next_u32() % 10 + __inl262_result: Optional[int] = None + __inl259_limit: int = 4294967290 + for __inl260_attempt in range(0, 32): + __inl261_draw: int = env.next_u32() + if __inl261_draw < __inl259_limit: + __inl262_result = __inl261_draw % 10 + if __inl262_result is not None: + break + if __inl262_result is None: + __inl262_result = env.next_u32() % 10 + __inl266_result: Optional[int] = None + __inl263_limit: int = 4294967290 + for __inl264_attempt in range(0, 32): + __inl265_draw: int = env.next_u32() + if __inl265_draw < __inl263_limit: + __inl266_result = __inl265_draw % 10 + if __inl266_result is not None: + break + if __inl266_result is None: + __inl266_result = env.next_u32() % 10 + __inl270_result: Optional[int] = None + __inl267_limit: int = 4294967290 + for __inl268_attempt in range(0, 32): + __inl269_draw: int = env.next_u32() + if __inl269_draw < __inl267_limit: + __inl270_result = __inl269_draw % 10 + if __inl270_result is not None: + break + if __inl270_result is None: + __inl270_result = env.next_u32() % 10 + __inl274_result: Optional[int] = None + __inl271_limit: int = 4294967290 + for __inl272_attempt in range(0, 32): + __inl273_draw: int = env.next_u32() + if __inl273_draw < __inl271_limit: + __inl274_result = __inl273_draw % 10 + if __inl274_result is not None: + break + if __inl274_result is None: + __inl274_result = env.next_u32() % 10 + __inl278_result: Optional[int] = None + __inl275_limit: int = 4294967290 + for __inl276_attempt in range(0, 32): + __inl277_draw: int = env.next_u32() + if __inl277_draw < __inl275_limit: + __inl278_result = __inl277_draw % 10 + if __inl278_result is not None: + break + if __inl278_result is None: + __inl278_result = env.next_u32() % 10 + base: str = ( + ( + ( + ( + ( + ( + ( + ( + ( + (str(__inl234_result) + str(__inl238_result)) + + str(__inl242_result) + ) + + str(__inl246_result) + ) + + str(__inl250_result) + ) + + str(__inl254_result) + ) + + str(__inl258_result) + ) + + str(__inl262_result) + ) + + str(__inl266_result) + ) + + str(__inl270_result) + ) + + str(__inl274_result) + ) + str(__inl278_result) + for attempt in range(0, 8): + __inl45_result: Optional[bool] = None + __inl42_first: int = ord(base[0]) + for __inl43_index in range(1, len(base)): + if ord(base[__inl43_index]) != __inl42_first: + __inl45_result = False + if __inl45_result is not None: + break + if __inl45_result is None: + __inl45_result = True + if not __inl45_result: + break + __inl282_result: Optional[int] = None + __inl279_limit: int = 4294967290 + for __inl280_attempt in range(0, 32): + __inl281_draw: int = env.next_u32() + if __inl281_draw < __inl279_limit: + __inl282_result = __inl281_draw % 10 + if __inl282_result is not None: + break + if __inl282_result is None: + __inl282_result = env.next_u32() % 10 + __inl286_result: Optional[int] = None + __inl283_limit: int = 4294967290 + for __inl284_attempt in range(0, 32): + __inl285_draw: int = env.next_u32() + if __inl285_draw < __inl283_limit: + __inl286_result = __inl285_draw % 10 + if __inl286_result is not None: + break + if __inl286_result is None: + __inl286_result = env.next_u32() % 10 + __inl290_result: Optional[int] = None + __inl287_limit: int = 4294967290 + for __inl288_attempt in range(0, 32): + __inl289_draw: int = env.next_u32() + if __inl289_draw < __inl287_limit: + __inl290_result = __inl289_draw % 10 + if __inl290_result is not None: + break + if __inl290_result is None: + __inl290_result = env.next_u32() % 10 + __inl294_result: Optional[int] = None + __inl291_limit: int = 4294967290 + for __inl292_attempt in range(0, 32): + __inl293_draw: int = env.next_u32() + if __inl293_draw < __inl291_limit: + __inl294_result = __inl293_draw % 10 + if __inl294_result is not None: + break + if __inl294_result is None: + __inl294_result = env.next_u32() % 10 + __inl298_result: Optional[int] = None + __inl295_limit: int = 4294967290 + for __inl296_attempt in range(0, 32): + __inl297_draw: int = env.next_u32() + if __inl297_draw < __inl295_limit: + __inl298_result = __inl297_draw % 10 + if __inl298_result is not None: + break + if __inl298_result is None: + __inl298_result = env.next_u32() % 10 + __inl302_result: Optional[int] = None + __inl299_limit: int = 4294967290 + for __inl300_attempt in range(0, 32): + __inl301_draw: int = env.next_u32() + if __inl301_draw < __inl299_limit: + __inl302_result = __inl301_draw % 10 + if __inl302_result is not None: + break + if __inl302_result is None: + __inl302_result = env.next_u32() % 10 + __inl306_result: Optional[int] = None + __inl303_limit: int = 4294967290 + for __inl304_attempt in range(0, 32): + __inl305_draw: int = env.next_u32() + if __inl305_draw < __inl303_limit: + __inl306_result = __inl305_draw % 10 + if __inl306_result is not None: + break + if __inl306_result is None: + __inl306_result = env.next_u32() % 10 + __inl310_result: Optional[int] = None + __inl307_limit: int = 4294967290 + for __inl308_attempt in range(0, 32): + __inl309_draw: int = env.next_u32() + if __inl309_draw < __inl307_limit: + __inl310_result = __inl309_draw % 10 + if __inl310_result is not None: + break + if __inl310_result is None: + __inl310_result = env.next_u32() % 10 + __inl314_result: Optional[int] = None + __inl311_limit: int = 4294967290 + for __inl312_attempt in range(0, 32): + __inl313_draw: int = env.next_u32() + if __inl313_draw < __inl311_limit: + __inl314_result = __inl313_draw % 10 + if __inl314_result is not None: + break + if __inl314_result is None: + __inl314_result = env.next_u32() % 10 + __inl318_result: Optional[int] = None + __inl315_limit: int = 4294967290 + for __inl316_attempt in range(0, 32): + __inl317_draw: int = env.next_u32() + if __inl317_draw < __inl315_limit: + __inl318_result = __inl317_draw % 10 + if __inl318_result is not None: + break + if __inl318_result is None: + __inl318_result = env.next_u32() % 10 + __inl322_result: Optional[int] = None + __inl319_limit: int = 4294967290 + for __inl320_attempt in range(0, 32): + __inl321_draw: int = env.next_u32() + if __inl321_draw < __inl319_limit: + __inl322_result = __inl321_draw % 10 + if __inl322_result is not None: + break + if __inl322_result is None: + __inl322_result = env.next_u32() % 10 + __inl326_result: Optional[int] = None + __inl323_limit: int = 4294967290 + for __inl324_attempt in range(0, 32): + __inl325_draw: int = env.next_u32() + if __inl325_draw < __inl323_limit: + __inl326_result = __inl325_draw % 10 + if __inl326_result is not None: + break + if __inl326_result is None: + __inl326_result = env.next_u32() % 10 + base = ( + ( + ( + ( + ( + ( + ( + ( + ( + ( + str(__inl282_result) + + str(__inl286_result) + ) + + str(__inl290_result) + ) + + str(__inl294_result) + ) + + str(__inl298_result) + ) + + str(__inl302_result) + ) + + str(__inl306_result) + ) + + str(__inl310_result) + ) + + str(__inl314_result) + ) + + str(__inl318_result) + ) + + str(__inl322_result) + ) + str(__inl326_result) + __inl49_cnpj: str = base + "00" + __inl46_sum: int = 0 + for __inl47_index in range(0, 12): + __inl46_sum = __inl46_sum + ( + (ord(__inl49_cnpj[__inl47_index]) - 48) + * [5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2][__inl47_index] + ) + __inl48_remainder: int = __inl46_sum % 11 + __inl51_result: int = 0 if (__inl48_remainder < 2) else (11 - __inl48_remainder) + first_digit: str = str(__inl51_result) + __inl55_cnpj: str = (base + first_digit) + "0" + __inl52_sum: int = 0 + for __inl53_index in range(0, 13): + __inl52_sum = __inl52_sum + ( + (ord(__inl55_cnpj[__inl53_index]) - 48) + * [6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2][__inl53_index] + ) + __inl54_remainder: int = __inl52_sum % 11 + __inl57_result: int = 0 if (__inl54_remainder < 2) else (11 - __inl54_remainder) + second_digit: str = str(__inl57_result) + return (base + first_digit) + second_digit diff --git a/core/out/python/generate_cpf.py b/core/out/python/generate_cpf.py new file mode 100644 index 000000000..e02f448f6 --- /dev/null +++ b/core/out/python/generate_cpf.py @@ -0,0 +1,280 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: generate-cpf +# content: 00fc978e8350 + +from typing import Optional +from ._support import Capabilities, DEFAULT_CAPABILITIES + +__all__ = ["generate_cpf", "generate_cpf_with"] + + +def generate_cpf() -> str: + """Generates a valid random CPF (Cadastro de Pessoas Físicas): 11 digits, under the check digit + rule (weights 10..2 and 11..2) the Receita Federal's Manual de Preenchimento da e-Financeira, + Anexo II specifies. + + Matches the published `generateCpf()` called with no state: a random 9-digit base — 8 digits + plus a região fiscal digit, also drawn at random here — redrawn while every digit of it is the + same, followed by its two check digits. The state code option is a DX concern: it only ever + picks which digit the 9th position draws from, never how the rest of the document is built. + """ + return generate_cpf_with(DEFAULT_CAPABILITIES) + + +def generate_cpf_with(env: Capabilities) -> str: + """`generate_cpf`, taking its capabilities explicitly. + + The public `generate_cpf` calls this with the platform's defaults. Pass your own to + supply a clock, a source of randomness or an HTTP client — which is what the + differential conformance driver does to make a run reproducible. + """ + __inl330_result: Optional[int] = None + __inl327_limit: int = 4294967290 + for __inl328_attempt in range(0, 32): + __inl329_draw: int = env.next_u32() + if __inl329_draw < __inl327_limit: + __inl330_result = __inl329_draw % 10 + if __inl330_result is not None: + break + if __inl330_result is None: + __inl330_result = env.next_u32() % 10 + __inl334_result: Optional[int] = None + __inl331_limit: int = 4294967290 + for __inl332_attempt in range(0, 32): + __inl333_draw: int = env.next_u32() + if __inl333_draw < __inl331_limit: + __inl334_result = __inl333_draw % 10 + if __inl334_result is not None: + break + if __inl334_result is None: + __inl334_result = env.next_u32() % 10 + __inl338_result: Optional[int] = None + __inl335_limit: int = 4294967290 + for __inl336_attempt in range(0, 32): + __inl337_draw: int = env.next_u32() + if __inl337_draw < __inl335_limit: + __inl338_result = __inl337_draw % 10 + if __inl338_result is not None: + break + if __inl338_result is None: + __inl338_result = env.next_u32() % 10 + __inl342_result: Optional[int] = None + __inl339_limit: int = 4294967290 + for __inl340_attempt in range(0, 32): + __inl341_draw: int = env.next_u32() + if __inl341_draw < __inl339_limit: + __inl342_result = __inl341_draw % 10 + if __inl342_result is not None: + break + if __inl342_result is None: + __inl342_result = env.next_u32() % 10 + __inl346_result: Optional[int] = None + __inl343_limit: int = 4294967290 + for __inl344_attempt in range(0, 32): + __inl345_draw: int = env.next_u32() + if __inl345_draw < __inl343_limit: + __inl346_result = __inl345_draw % 10 + if __inl346_result is not None: + break + if __inl346_result is None: + __inl346_result = env.next_u32() % 10 + __inl350_result: Optional[int] = None + __inl347_limit: int = 4294967290 + for __inl348_attempt in range(0, 32): + __inl349_draw: int = env.next_u32() + if __inl349_draw < __inl347_limit: + __inl350_result = __inl349_draw % 10 + if __inl350_result is not None: + break + if __inl350_result is None: + __inl350_result = env.next_u32() % 10 + __inl354_result: Optional[int] = None + __inl351_limit: int = 4294967290 + for __inl352_attempt in range(0, 32): + __inl353_draw: int = env.next_u32() + if __inl353_draw < __inl351_limit: + __inl354_result = __inl353_draw % 10 + if __inl354_result is not None: + break + if __inl354_result is None: + __inl354_result = env.next_u32() % 10 + __inl358_result: Optional[int] = None + __inl355_limit: int = 4294967290 + for __inl356_attempt in range(0, 32): + __inl357_draw: int = env.next_u32() + if __inl357_draw < __inl355_limit: + __inl358_result = __inl357_draw % 10 + if __inl358_result is not None: + break + if __inl358_result is None: + __inl358_result = env.next_u32() % 10 + __inl362_result: Optional[int] = None + __inl359_limit: int = 4294967290 + for __inl360_attempt in range(0, 32): + __inl361_draw: int = env.next_u32() + if __inl361_draw < __inl359_limit: + __inl362_result = __inl361_draw % 10 + if __inl362_result is not None: + break + if __inl362_result is None: + __inl362_result = env.next_u32() % 10 + base: str = ( + ( + ( + ( + ( + ( + (str(__inl330_result) + str(__inl334_result)) + + str(__inl338_result) + ) + + str(__inl342_result) + ) + + str(__inl346_result) + ) + + str(__inl350_result) + ) + + str(__inl354_result) + ) + + str(__inl358_result) + ) + str(__inl362_result) + for attempt in range(0, 8): + __inl61_result: Optional[bool] = None + __inl58_first: int = ord(base[0]) + for __inl59_index in range(1, len(base)): + if ord(base[__inl59_index]) != __inl58_first: + __inl61_result = False + if __inl61_result is not None: + break + if __inl61_result is None: + __inl61_result = True + if not __inl61_result: + break + __inl366_result: Optional[int] = None + __inl363_limit: int = 4294967290 + for __inl364_attempt in range(0, 32): + __inl365_draw: int = env.next_u32() + if __inl365_draw < __inl363_limit: + __inl366_result = __inl365_draw % 10 + if __inl366_result is not None: + break + if __inl366_result is None: + __inl366_result = env.next_u32() % 10 + __inl370_result: Optional[int] = None + __inl367_limit: int = 4294967290 + for __inl368_attempt in range(0, 32): + __inl369_draw: int = env.next_u32() + if __inl369_draw < __inl367_limit: + __inl370_result = __inl369_draw % 10 + if __inl370_result is not None: + break + if __inl370_result is None: + __inl370_result = env.next_u32() % 10 + __inl374_result: Optional[int] = None + __inl371_limit: int = 4294967290 + for __inl372_attempt in range(0, 32): + __inl373_draw: int = env.next_u32() + if __inl373_draw < __inl371_limit: + __inl374_result = __inl373_draw % 10 + if __inl374_result is not None: + break + if __inl374_result is None: + __inl374_result = env.next_u32() % 10 + __inl378_result: Optional[int] = None + __inl375_limit: int = 4294967290 + for __inl376_attempt in range(0, 32): + __inl377_draw: int = env.next_u32() + if __inl377_draw < __inl375_limit: + __inl378_result = __inl377_draw % 10 + if __inl378_result is not None: + break + if __inl378_result is None: + __inl378_result = env.next_u32() % 10 + __inl382_result: Optional[int] = None + __inl379_limit: int = 4294967290 + for __inl380_attempt in range(0, 32): + __inl381_draw: int = env.next_u32() + if __inl381_draw < __inl379_limit: + __inl382_result = __inl381_draw % 10 + if __inl382_result is not None: + break + if __inl382_result is None: + __inl382_result = env.next_u32() % 10 + __inl386_result: Optional[int] = None + __inl383_limit: int = 4294967290 + for __inl384_attempt in range(0, 32): + __inl385_draw: int = env.next_u32() + if __inl385_draw < __inl383_limit: + __inl386_result = __inl385_draw % 10 + if __inl386_result is not None: + break + if __inl386_result is None: + __inl386_result = env.next_u32() % 10 + __inl390_result: Optional[int] = None + __inl387_limit: int = 4294967290 + for __inl388_attempt in range(0, 32): + __inl389_draw: int = env.next_u32() + if __inl389_draw < __inl387_limit: + __inl390_result = __inl389_draw % 10 + if __inl390_result is not None: + break + if __inl390_result is None: + __inl390_result = env.next_u32() % 10 + __inl394_result: Optional[int] = None + __inl391_limit: int = 4294967290 + for __inl392_attempt in range(0, 32): + __inl393_draw: int = env.next_u32() + if __inl393_draw < __inl391_limit: + __inl394_result = __inl393_draw % 10 + if __inl394_result is not None: + break + if __inl394_result is None: + __inl394_result = env.next_u32() % 10 + __inl398_result: Optional[int] = None + __inl395_limit: int = 4294967290 + for __inl396_attempt in range(0, 32): + __inl397_draw: int = env.next_u32() + if __inl397_draw < __inl395_limit: + __inl398_result = __inl397_draw % 10 + if __inl398_result is not None: + break + if __inl398_result is None: + __inl398_result = env.next_u32() % 10 + base = ( + ( + ( + ( + ( + ( + (str(__inl366_result) + str(__inl370_result)) + + str(__inl374_result) + ) + + str(__inl378_result) + ) + + str(__inl382_result) + ) + + str(__inl386_result) + ) + + str(__inl390_result) + ) + + str(__inl394_result) + ) + str(__inl398_result) + __inl65_cpf: str = base + "00" + __inl62_sum: int = 0 + for __inl63_index in range(0, 9): + __inl62_sum = __inl62_sum + ( + (ord(__inl65_cpf[__inl63_index]) - 48) * (10 - __inl63_index) + ) + __inl64_remainder: int = __inl62_sum % 11 + __inl66_result: int = 0 if (__inl64_remainder < 2) else (11 - __inl64_remainder) + first_digit: str = str(__inl66_result) + __inl70_cpf: str = (base + first_digit) + "0" + __inl67_sum: int = 0 + for __inl68_index in range(0, 10): + __inl67_sum = __inl67_sum + ( + (ord(__inl70_cpf[__inl68_index]) - 48) * (11 - __inl68_index) + ) + __inl69_remainder: int = __inl67_sum % 11 + __inl71_result: int = 0 if (__inl69_remainder < 2) else (11 - __inl69_remainder) + second_digit: str = str(__inl71_result) + return (base + first_digit) + second_digit diff --git a/core/out/python/get_address_info_by_cep.py b/core/out/python/get_address_info_by_cep.py new file mode 100644 index 000000000..493994e92 --- /dev/null +++ b/core/out/python/get_address_info_by_cep.py @@ -0,0 +1,154 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: get-address-info-by-cep +# content: daa7ccc2e0b6 + +from typing import Optional +from dataclasses import dataclass +from ._support import race_first_some +import re +from .lib.json import json_string_field +from .errors import GetAddressInfoByCepNotFoundError, GetAddressInfoByCepValidationError +from ._support import Capabilities, DEFAULT_CAPABILITIES, HttpRequest, HttpResponse + +__all__ = ["get_address_info_by_cep", "get_address_info_by_cep_with"] + + +@dataclass(frozen=True) +class AddressInfo: + cep: str + state: str + city: str + neighborhood: str + street: str + + +_GET_ADDRESS_INFO_BY_CEP_PATTERN_1 = re.compile("[^0-9]") + +_GET_ADDRESS_INFO_BY_CEP_PATTERN_2 = re.compile("[0-9]{8}") + + +def _get_with_retry(url: str, env: Capabilities) -> Optional[HttpResponse]: + """One GET, retried the way the published package retries: twice more, 250 ms apart.""" + for attempt in range(0, 3): + if attempt > 0: + env.sleep(250) + response: Optional[HttpResponse] = env.request( + HttpRequest( + method="GET", url=url, headers=[], body="", timeout_millis=10000 + ) + ) + if response is not None: + return response + return None + + +def _is_ok(status: int) -> bool: + """Whether the status is a 2xx.""" + return (status >= 200) and (status < 300) + + +def _fetch_via_cep(cep: str, env: Capabilities) -> Optional[AddressInfo]: + """ViaCEP answers a JSON object, and marks an unknown CEP with `"erro"`.""" + response: Optional[HttpResponse] = _get_with_retry( + (("https://viacep.com.br/ws/" + cep) + "/json/"), env + ) + if (response is None) or (not _is_ok(response.status)): + return None + code: str = ( + __value + if (__value := json_string_field(response.body, "cep")) is not None + else "" + ) + if code == "": + return None + return AddressInfo( + cep=_GET_ADDRESS_INFO_BY_CEP_PATTERN_1.sub("", code), + state=( + __value + if (__value := json_string_field(response.body, "uf")) is not None + else "" + ), + city=( + __value + if (__value := json_string_field(response.body, "localidade")) is not None + else "" + ), + neighborhood=( + __value + if (__value := json_string_field(response.body, "bairro")) is not None + else "" + ), + street=( + __value + if (__value := json_string_field(response.body, "logradouro")) is not None + else "" + ), + ) + + +def _fetch_brasil_api(cep: str, env: Capabilities) -> Optional[AddressInfo]: + """BrasilAPI answers 404 for an unknown CEP.""" + response: Optional[HttpResponse] = _get_with_retry( + ("https://brasilapi.com.br/api/cep/v1/" + cep), env + ) + if (response is None) or (not _is_ok(response.status)): + return None + code: str = ( + __value + if (__value := json_string_field(response.body, "cep")) is not None + else "" + ) + if code == "": + return None + return AddressInfo( + cep=_GET_ADDRESS_INFO_BY_CEP_PATTERN_1.sub("", code), + state=( + __value + if (__value := json_string_field(response.body, "state")) is not None + else "" + ), + city=( + __value + if (__value := json_string_field(response.body, "city")) is not None + else "" + ), + neighborhood=( + __value + if (__value := json_string_field(response.body, "neighborhood")) is not None + else "" + ), + street=( + __value + if (__value := json_string_field(response.body, "street")) is not None + else "" + ), + ) + + +def get_address_info_by_cep(cep: str) -> AddressInfo: + """The address of a CEP, from the first service that answers. + + The two services are queried concurrently and the first answer wins; the losing request may + still finish, and its answer is dropped, which is why only idempotent GETs belong here. Each + request is retried twice, 250 ms apart, exactly as the published package does. Turning a host + value into the 8 digits this takes is the DX's job. + """ + return get_address_info_by_cep_with(cep, DEFAULT_CAPABILITIES) + + +def get_address_info_by_cep_with(cep: str, env: Capabilities) -> AddressInfo: + """`get_address_info_by_cep`, taking its capabilities explicitly. + + The public `get_address_info_by_cep` calls this with the platform's defaults. Pass your own to + supply a clock, a source of randomness or an HTTP client — which is what the + differential conformance driver does to make a run reproducible. + """ + if not (_GET_ADDRESS_INFO_BY_CEP_PATTERN_2.fullmatch(cep) is not None): + raise GetAddressInfoByCepValidationError("CEP inv\u00e1lido") + address: Optional[AddressInfo] = race_first_some( + [(lambda: _fetch_via_cep(cep, env)), (lambda: _fetch_brasil_api(cep, env))] + ) + if address is None: + raise GetAddressInfoByCepNotFoundError("CEP n\u00e3o encontrado") + return address diff --git a/core/out/python/get_holidays.py b/core/out/python/get_holidays.py new file mode 100644 index 000000000..fc0f79692 --- /dev/null +++ b/core/out/python/get_holidays.py @@ -0,0 +1,178 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: get-holidays +# content: b4d687034433 + +from typing import List, Literal +from dataclasses import dataclass +from .lib.civil import civil_date_10 +from .std.date import ymd_to_days + +__all__ = ["get_holidays"] + + +@dataclass(frozen=True) +class Holiday: + name: str + date: int + type: Literal["national", "optional", "religious", "state"] + + +def get_holidays(year: int) -> List[Holiday]: + """The Brazilian national holidays of a year, sorted by date. + + The order is the one the published package produces: the fixed holidays in statutory order, + then the Easter-derived ones, sorted by date with a stable sort, so two holidays on the same + day keep the order they were built in. State holidays are not part of this pilot. + """ + holidays: List[Holiday] = [] + holidays.append( + Holiday( + name="Ano novo", + date=( + __value + if (__value := ymd_to_days(year, 1, 1)) is not None + else min(max(0, -719162), 2932896) + ), + type="national", + ) + ) + holidays.append( + Holiday( + name="Tiradentes", + date=( + __value + if (__value := ymd_to_days(year, 4, 21)) is not None + else min(max(0, -719162), 2932896) + ), + type="national", + ) + ) + holidays.append( + Holiday( + name="Dia do trabalhador", + date=( + __value + if (__value := ymd_to_days(year, 5, 1)) is not None + else min(max(0, -719162), 2932896) + ), + type="national", + ) + ) + holidays.append( + Holiday( + name="Independ\u00eancia do Brasil", + date=( + __value + if (__value := ymd_to_days(year, 9, 7)) is not None + else min(max(0, -719162), 2932896) + ), + type="national", + ) + ) + holidays.append( + Holiday( + name="Nossa Senhora Aparecida", + date=( + __value + if (__value := ymd_to_days(year, 10, 12)) is not None + else min(max(0, -719162), 2932896) + ), + type="national", + ) + ) + holidays.append( + Holiday( + name="Finados", + date=( + __value + if (__value := ymd_to_days(year, 11, 2)) is not None + else min(max(0, -719162), 2932896) + ), + type="national", + ) + ) + holidays.append( + Holiday( + name="Proclama\u00e7\u00e3o da Rep\u00fablica", + date=( + __value + if (__value := ymd_to_days(year, 11, 15)) is not None + else min(max(0, -719162), 2932896) + ), + type="national", + ) + ) + holidays.append( + Holiday( + name="Natal", + date=( + __value + if (__value := ymd_to_days(year, 12, 25)) is not None + else min(max(0, -719162), 2932896) + ), + type="national", + ) + ) + if year >= 2024: + holidays.append( + Holiday( + name="Dia da Consci\u00eancia Negra", + date=( + __value + if (__value := ymd_to_days(year, 11, 20)) is not None + else min(max(0, -719162), 2932896) + ), + type="national", + ) + ) + __inl217_a: int = year % 19 + __inl218_b: int = year // 100 + __inl219_c: int = year % 100 + __inl220_d: int = __inl218_b // 4 + __inl221_e: int = __inl218_b % 4 + __inl222_h: int = (((((19 * __inl217_a) + __inl218_b) - __inl220_d) - 6) + 15) % 30 + __inl223_i: int = __inl219_c // 4 + __inl224_k: int = __inl219_c % 4 + __inl225_l: int = ( + (((32 + (2 * __inl221_e)) + (2 * __inl223_i)) - __inl222_h) - __inl224_k + ) % 7 + __inl226_m: int = ((__inl217_a + (11 * __inl222_h)) + (22 * __inl225_l)) // 451 + __inl227_day: int = ((__inl222_h + __inl225_l) - (7 * __inl226_m)) + 114 + __inl229_result: int = min( + max((((__inl227_day % 31) + 1) + (((__inl227_day // 31) - 3) * 31)), 22), 56 + ) + __inl102_day_of_march: int = __inl229_result + __inl104_result: int = ( + ( + __value + if (__value := ymd_to_days(year, 3, __inl102_day_of_march)) is not None + else min(max(0, -719162), 2932896) + ) + if (__inl102_day_of_march <= 31) + else civil_date_10(year, (__inl102_day_of_march - 31)) + ) + easter: int = __inl104_result + holidays.append( + Holiday( + name="Carnaval (ter\u00e7a-feira)", + date=(easter + -47 if -719162 <= easter + -47 <= 2932896 else easter), + type="optional", + ) + ) + holidays.append( + Holiday( + name="Sexta-feira Santa", + date=(easter + -2 if -719162 <= easter + -2 <= 2932896 else easter), + type="national", + ) + ) + holidays.append(Holiday(name="P\u00e1scoa", date=easter, type="religious")) + holidays.append( + Holiday( + name="Corpus Christi", + date=(easter + 60 if -719162 <= easter + 60 <= 2932896 else easter), + type="optional", + ) + ) + return sorted(holidays, key=(lambda holiday: holiday.date)) diff --git a/core/out/python/is_business_day.py b/core/out/python/is_business_day.py new file mode 100644 index 000000000..9287cdea2 --- /dev/null +++ b/core/out/python/is_business_day.py @@ -0,0 +1,33 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: is-business-day +# content: 3605fae387a8 + +from .get_holidays import get_holidays +from .std.date import year_from_days + +__all__ = ["is_business_day"] + + +def is_business_day(value: int, include_optional: bool) -> bool: + """Whether a date is a Brazilian business day (dia útil). + + A day is not a business day when it falls on a weekend, or when it is one of the holidays + `getHolidays` lists for its year. `includeOptional` decides whether the ponto facultativo + entries (Carnaval, Corpus Christi) count; the published package defaults it to `true`, and + supplying that default is the DX's job. + + Only the years 1900 to 2099 are supported, the range the holiday rules are stated for. + """ + year: int = year_from_days(value) + if (year < 1900) or (year > 2099): + return False + weekday: int = ((value + 3) % 7) + 1 + if (weekday == 6) or (weekday == 7): + return False + for holiday in get_holidays(year): + if (not include_optional) and (holiday.type == "optional"): + continue + if (-1 if holiday.date < value else (1 if holiday.date > value else 0)) == 0: + return False + return True diff --git a/core/out/python/is_valid_cnpj.py b/core/out/python/is_valid_cnpj.py new file mode 100644 index 000000000..bbac00b0b --- /dev/null +++ b/core/out/python/is_valid_cnpj.py @@ -0,0 +1,85 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: is-valid-cnpj +# content: 611130f5f12f + +from typing import List, Literal, Optional +import re +from .lib.cnpj import cnpj_check_digit, is_repeated_cnpj + +__all__ = ["is_valid_cnpj"] + +IS_VALID_CNPJ_TABLE_1: List[int] = [5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2] + +IS_VALID_CNPJ_TABLE_2: List[int] = [6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2] + +_IS_VALID_CNPJ_PATTERN_1 = re.compile("[^0-9A-Za-z]") + +_IS_VALID_CNPJ_PATTERN_2 = re.compile( + "[0-9A-Z]{2}[\\x09-\\x0d \\--/\\u00a0\\u1680\\u2000-\\u200a\\u2028-\\u2029\\u202f\\u205f\\u3000\\ufeff]*[0-9A-Z]{3}[\\x09-\\x0d \\--/\\u00a0\\u1680\\u2000-\\u200a\\u2028-\\u2029\\u202f\\u205f\\u3000\\ufeff]*[0-9A-Z]{3}[\\x09-\\x0d \\--/\\u00a0\\u1680\\u2000-\\u200a\\u2028-\\u2029\\u202f\\u205f\\u3000\\ufeff]*[0-9A-Z]{4}[\\x09-\\x0d \\--/\\u00a0\\u1680\\u2000-\\u200a\\u2028-\\u2029\\u202f\\u205f\\u3000\\ufeff]*[0-9]{2}" +) + +_IS_VALID_CNPJ_PATTERN_3 = re.compile("[a-z]") + +_IS_VALID_CNPJ_PATTERN_4 = re.compile("[^0-9]") + +_IS_VALID_CNPJ_PATTERN_5 = re.compile( + "[0-9]{2}[\\x09-\\x0d \\--/\\u00a0\\u1680\\u2000-\\u200a\\u2028-\\u2029\\u202f\\u205f\\u3000\\ufeff]*[0-9]{3}[\\x09-\\x0d \\--/\\u00a0\\u1680\\u2000-\\u200a\\u2028-\\u2029\\u202f\\u205f\\u3000\\ufeff]*[0-9]{3}[\\x09-\\x0d \\--/\\u00a0\\u1680\\u2000-\\u200a\\u2028-\\u2029\\u202f\\u205f\\u3000\\ufeff]*[0-9]{4}[\\x09-\\x0d \\--/\\u00a0\\u1680\\u2000-\\u200a\\u2028-\\u2029\\u202f\\u205f\\u3000\\ufeff]*[0-9]{2}" +) + + +def is_valid_cnpj(cnpj: str, version: Literal["1", "2"]) -> bool: + """Validates a CNPJ (Cadastro Nacional da Pessoa Jurídica), numeric or alphanumeric. + + Version `"2"` accepts the alphanumeric format as well; a value with no letters is always read + as the numeric one, which is also where the reserved repeated numbers are rejected. Mapping a + missing or unexpected `options.version` onto `"1"` is the DX's job. + """ + trimmed: str = cnpj.strip( + "\t\n\u000b\u000c\r \u00a0\u1680\u2000\u2001\u2002\u2003\u2004\u2005\u2006\u2007\u2008\u2009\u200a\u2028\u2029\u202f\u205f\u3000\ufeff" + ) + if version == "2": + cleaned: str = _IS_VALID_CNPJ_PATTERN_1.sub("", cnpj).upper() + __inl114_result: Optional[bool] = None + for __inl111_index in range(0, len(cleaned)): + __inl112_point: int = ( + ord(cleaned[__inl111_index]) + if 0 <= __inl111_index < len(cleaned) + else 0 + ) + if (__inl112_point >= 65) and (__inl112_point <= 90): + __inl114_result = True + if __inl114_result is not None: + break + if __inl114_result is None: + __inl114_result = False + if __inl114_result and (len(cleaned) == 14): + return ( + _IS_VALID_CNPJ_PATTERN_2.fullmatch( + _IS_VALID_CNPJ_PATTERN_3.sub( + lambda match: match.group().upper(), trimmed + ) + ) + is not None + ) and ( + ( + (ord(cleaned[12]) - 48) + == cnpj_check_digit(cleaned, IS_VALID_CNPJ_TABLE_1) + ) + and ( + (ord(cleaned[13]) - 48) + == cnpj_check_digit(cleaned, IS_VALID_CNPJ_TABLE_2) + ) + ) + numeric: str = _IS_VALID_CNPJ_PATTERN_4.sub("", cnpj) + if len(numeric) != 14: + return False + return ( + (_IS_VALID_CNPJ_PATTERN_5.fullmatch(trimmed) is not None) + and (not is_repeated_cnpj(numeric)) + ) and ( + ((ord(numeric[12]) - 48) == cnpj_check_digit(numeric, IS_VALID_CNPJ_TABLE_1)) + and ( + (ord(numeric[13]) - 48) == cnpj_check_digit(numeric, IS_VALID_CNPJ_TABLE_2) + ) + ) diff --git a/core/out/python/is_valid_cpf.py b/core/out/python/is_valid_cpf.py new file mode 100644 index 000000000..6a6ee9bb5 --- /dev/null +++ b/core/out/python/is_valid_cpf.py @@ -0,0 +1,57 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: is-valid-cpf +# content: dae364e11fb3 + +from typing import Optional +import re +from .lib.cpf import cpf_check_digit_1 + +__all__ = ["is_valid_cpf"] + +_IS_VALID_CPF_PATTERN_1 = re.compile( + "[0-9]{3}[\\x09-\\x0d \\--/\\u00a0\\u1680\\u2000-\\u200a\\u2028-\\u2029\\u202f\\u205f\\u3000\\ufeff]*[0-9]{3}[\\x09-\\x0d \\--/\\u00a0\\u1680\\u2000-\\u200a\\u2028-\\u2029\\u202f\\u205f\\u3000\\ufeff]*[0-9]{3}[\\x09-\\x0d \\--/\\u00a0\\u1680\\u2000-\\u200a\\u2028-\\u2029\\u202f\\u205f\\u3000\\ufeff]*[0-9]{2}" +) + +_IS_VALID_CPF_PATTERN_2 = re.compile("[^0-9]") + + +def is_valid_cpf(cpf: str) -> bool: + """Validates a CPF (Cadastro de Pessoas Físicas). + + The core takes the value as written, accepting the usual mask characters; turning a host value + into a string is the DX's job. + """ + if not ( + _IS_VALID_CPF_PATTERN_1.fullmatch( + cpf.strip( + "\t\n\u000b\u000c\r \u00a0\u1680\u2000\u2001\u2002\u2003\u2004\u2005\u2006\u2007\u2008\u2009\u200a\u2028\u2029\u202f\u205f\u3000\ufeff" + ) + ) + is not None + ): + return False + digits: str = _IS_VALID_CPF_PATTERN_2.sub("", cpf) + if len(digits) != 11: + return False + __inl118_result: Optional[bool] = None + __inl115_first: int = ord(digits[0]) + for __inl116_index in range(1, 11): + if ord(digits[__inl116_index]) != __inl115_first: + __inl118_result = False + if __inl118_result is not None: + break + if __inl118_result is None: + __inl118_result = True + if __inl118_result: + return False + __inl119_sum: int = 0 + for __inl120_index in range(0, 9): + __inl119_sum = __inl119_sum + ( + (ord(digits[__inl120_index]) - 48) * (10 - __inl120_index) + ) + __inl121_remainder: int = __inl119_sum % 11 + __inl123_result: int = 0 if (__inl121_remainder < 2) else (11 - __inl121_remainder) + return ((ord(digits[9]) - 48) == __inl123_result) and ( + (ord(digits[10]) - 48) == cpf_check_digit_1(digits) + ) diff --git a/core/out/python/lib/__init__.py b/core/out/python/lib/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/core/out/python/lib/civil.py b/core/out/python/lib/civil.py new file mode 100644 index 000000000..9306787bd --- /dev/null +++ b/core/out/python/lib/civil.py @@ -0,0 +1,17 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: lib/civil +# content: c2aca1eb50dc + +from ..std.date import ymd_to_days + +__all__ = ["civil_date_10"] + + +def civil_date_10(year: int, day: int) -> int: + """A fixed day of a year, with the unreachable fallback named once.""" + return ( + __value + if (__value := ymd_to_days(year, 4, day)) is not None + else min(max(0, -719162), 2932896) + ) diff --git a/core/out/python/lib/cnpj.py b/core/out/python/lib/cnpj.py new file mode 100644 index 000000000..fa8ce26ac --- /dev/null +++ b/core/out/python/lib/cnpj.py @@ -0,0 +1,26 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: lib/cnpj +# content: 78a3fa4783b9 + +from typing import List + +__all__ = ["cnpj_check_digit", "is_repeated_cnpj"] + + +def cnpj_check_digit(cnpj: str, weights: List[int]) -> int: + """The check digit of a CNPJ base, under the rule both versions share.""" + sum: int = 0 + for index in range(0, len(weights)): + sum = sum + ((ord(cnpj[index]) - 48) * weights[index]) + remainder: int = sum % 11 + return 0 if (remainder < 2) else (11 - remainder) + + +def is_repeated_cnpj(value: str) -> bool: + """Whether every character of a 14 character value is the same one.""" + first: int = ord(value[0]) + for index in range(1, 14): + if ord(value[index]) != first: + return False + return True diff --git a/core/out/python/lib/cpf.py b/core/out/python/lib/cpf.py new file mode 100644 index 000000000..fd32a013f --- /dev/null +++ b/core/out/python/lib/cpf.py @@ -0,0 +1,16 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: lib/cpf +# content: 5b319831cfe4 + + +__all__ = ["cpf_check_digit_1"] + + +def cpf_check_digit_1(cpf: str) -> int: + """The check digit of a CPF base, under the Receita Federal rule (weights 10..2 and 11..2).""" + sum: int = 0 + for index in range(0, 10): + sum = sum + ((ord(cpf[index]) - 48) * (11 - index)) + remainder: int = sum % 11 + return 0 if (remainder < 2) else (11 - remainder) diff --git a/core/out/python/lib/format.py b/core/out/python/lib/format.py new file mode 100644 index 000000000..8006e4163 --- /dev/null +++ b/core/out/python/lib/format.py @@ -0,0 +1,39 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: lib/format +# content: 0038a81c2e4d + + +__all__ = ["pattern_slots", "format_with_pattern"] + + +def pattern_slots(pattern: str) -> int: + """How many scalars of the value a pattern consumes.""" + slots: int = 0 + for index in range(0, len(pattern)): + symbol: str = pattern[index] if 0 <= index < len(pattern) else "" + if (symbol == "0") or (symbol == "*"): + slots = slots + 1 + return slots + + +def format_with_pattern(value: str, pattern: str, pad: bool) -> str: + """Formats a value against a pattern, optionally left padding it with zeros first.""" + padded: str = value.rjust(pattern_slots(pattern), "0") if pad else value + out: str = "" + taken: int = 0 + for index in range(0, len(pattern)): + symbol: str = pattern[index] if 0 <= index < len(pattern) else "" + if (symbol == "0") or (symbol == "*"): + if taken >= len(padded): + return out + out = out + ( + "*" + if (symbol == "*") + else (padded[taken] if 0 <= taken < len(padded) else "") + ) + taken = taken + 1 + else: + if taken < len(padded): + out = out + symbol + return out diff --git a/core/out/python/lib/json.py b/core/out/python/lib/json.py new file mode 100644 index 000000000..f7f117cf7 --- /dev/null +++ b/core/out/python/lib/json.py @@ -0,0 +1,111 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: lib/json +# content: 02cc75dfd626 + +from typing import List, Optional + +__all__ = ["json_string_field"] + + +def json_string_field(body: str, key: str) -> Optional[str]: + """The string value of a top-level JSON field, or absent when the field is missing or is not a + string. Escapes are decoded; a surrogate pair is left as its two escaped halves, which no CEP + provider emits. + """ + points: List[int] = [ord(__c) for __c in body] + needle: List[int] = [ord(__c) for __c in (('"' + key) + '"')] + for index in range(0, len(points)): + __inl76_result: Optional[bool] = None + for __inl72_offset in range(0, len(needle)): + if ( + points[(index + __inl72_offset)] + if 0 <= (index + __inl72_offset) < len(points) + else -1 + ) != (needle[__inl72_offset] if 0 <= __inl72_offset < len(needle) else -2): + __inl76_result = False + if __inl76_result is not None: + break + if __inl76_result is None: + __inl76_result = True + if not __inl76_result: + continue + cursor: int = index + len(needle) + for skip in range(0, 8): + __inl77_point: int = points[cursor] if 0 <= cursor < len(points) else 0 + if ( + ((__inl77_point == 32) or (__inl77_point == 9)) or (__inl77_point == 10) + ) or (__inl77_point == 13): + cursor = cursor + 1 + if (points[cursor] if 0 <= cursor < len(points) else 0) != 58: + continue + cursor = cursor + 1 + for skip in range(0, 8): + __inl78_point: int = points[cursor] if 0 <= cursor < len(points) else 0 + if ( + ((__inl78_point == 32) or (__inl78_point == 9)) or (__inl78_point == 10) + ) or (__inl78_point == 13): + cursor = cursor + 1 + if (points[cursor] if 0 <= cursor < len(points) else 0) != 34: + continue + cursor = cursor + 1 + out: List[int] = [] + for step in range(0, len(points)): + point: int = points[cursor] if 0 <= cursor < len(points) else -1 + if (point == -1) or (point == 34): + return "".join(chr(__p) for __p in out) + if point == 92: + escaped: int = ( + points[(cursor + 1)] if 0 <= (cursor + 1) < len(points) else -1 + ) + if escaped == 110: + out.append(10) + cursor = cursor + 2 + else: + if escaped == 116: + out.append(9) + cursor = cursor + 2 + else: + if escaped == 114: + out.append(13) + cursor = cursor + 2 + else: + if escaped == 117: + __inl84_start: int = cursor + 2 + __inl79_value: int = 0 + for __inl80_offset in range(0, 4): + __inl81_point: int = ( + points[(__inl84_start + __inl80_offset)] + if 0 + <= (__inl84_start + __inl80_offset) + < len(points) + else 48 + ) + __inl82_digit: int = 0 + if (__inl81_point >= 48) and (__inl81_point <= 57): + __inl82_digit = __inl81_point - 48 + else: + if (__inl81_point >= 97) and ( + __inl81_point <= 102 + ): + __inl82_digit = __inl81_point - 87 + else: + if (__inl81_point >= 65) and ( + __inl81_point <= 70 + ): + __inl82_digit = __inl81_point - 55 + __inl79_value = (__inl79_value * 16) + __inl82_digit + __inl85_result: int = min(__inl79_value, 65535) + out.append(__inl85_result) + cursor = cursor + 6 + else: + if escaped >= 0: + out.append(escaped) + cursor = cursor + 2 + else: + cursor = cursor + 1 + else: + out.append(point) + cursor = cursor + 1 + return "".join(chr(__p) for __p in out) + return None diff --git a/core/out/python/std/__init__.py b/core/out/python/std/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/core/out/python/std/date.py b/core/out/python/std/date.py new file mode 100644 index 000000000..dd2679c97 --- /dev/null +++ b/core/out/python/std/date.py @@ -0,0 +1,446 @@ +# Code generated by the logic engine. DO NOT EDIT. +# engine: 0.1.0 +# source: std/date +# content: 8041c981a090 + +from typing import Optional + +__all__ = ["month_from_days", "day_from_days", "ymd_to_days", "year_from_days"] + + +def month_from_days(days: int) -> int: + """The month of a date given as days since 1970-01-01.""" + shifted: int = days + 719468 + __inl6_result: Optional[int] = None + __inl4_quotient: int = shifted // 146097 + if (shifted < 0) and ((__inl4_quotient * 146097) != shifted): + __inl6_result = __inl4_quotient - 1 + if __inl6_result is None: + __inl6_result = __inl4_quotient + era: int = __inl6_result + day_of_era: int = shifted - (era * 146097) + year_of_era: int = ( + -(abs(__td_a) // abs(__td_b)) + if ( + ( + __td_a := ( + ( + ( + day_of_era + - ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := day_of_era), + (__td_b := 1460), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + + ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := day_of_era), + (__td_b := 36524), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + - ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := day_of_era), + (__td_b := 146096), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + ), + (__td_b := 365), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + day_of_year: int = day_of_era - ( + ( + (365 * year_of_era) + + ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := year_of_era), + (__td_b := 4), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + - ( + -(abs(__td_a) // abs(__td_b)) + if ((__td_a := year_of_era), (__td_b := 100), (__td_a < 0) != (__td_b < 0))[ + 2 + ] + else abs(__td_a) // abs(__td_b) + ) + ) + month_prime: int = ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := ((5 * day_of_year) + 2)), + (__td_b := 153), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + return min( + max(((month_prime + 3) if (month_prime < 10) else (month_prime - 9)), 1), 12 + ) + + +def day_from_days(days: int) -> int: + """The day of month of a date given as days since 1970-01-01.""" + shifted: int = days + 719468 + __inl9_result: Optional[int] = None + __inl7_quotient: int = shifted // 146097 + if (shifted < 0) and ((__inl7_quotient * 146097) != shifted): + __inl9_result = __inl7_quotient - 1 + if __inl9_result is None: + __inl9_result = __inl7_quotient + era: int = __inl9_result + day_of_era: int = shifted - (era * 146097) + year_of_era: int = ( + -(abs(__td_a) // abs(__td_b)) + if ( + ( + __td_a := ( + ( + ( + day_of_era + - ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := day_of_era), + (__td_b := 1460), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + + ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := day_of_era), + (__td_b := 36524), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + - ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := day_of_era), + (__td_b := 146096), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + ), + (__td_b := 365), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + day_of_year: int = day_of_era - ( + ( + (365 * year_of_era) + + ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := year_of_era), + (__td_b := 4), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + - ( + -(abs(__td_a) // abs(__td_b)) + if ((__td_a := year_of_era), (__td_b := 100), (__td_a < 0) != (__td_b < 0))[ + 2 + ] + else abs(__td_a) // abs(__td_b) + ) + ) + month_prime: int = ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := ((5 * day_of_year) + 2)), + (__td_b := 153), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + return min( + max( + ( + ( + day_of_year + - ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := ((153 * month_prime) + 2)), + (__td_b := 5), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + + 1 + ), + 1, + ), + 31, + ) + + +def ymd_to_days(year: int, month: int, day: int) -> Optional[int]: + """Days since 1970-01-01, or absent when the components do not name a real date. + + The bounds are checked here rather than in a helper because the checker reads a guard, not a + called predicate: after this `if`, the three components carry the ranges `daysFromCivil` + requires, and the round trip rejects a day the month does not have. + """ + if ( + ((((year < 1) or (year > 9999)) or (month < 1)) or (month > 12)) or (day < 1) + ) or (day > 31): + return None + __inl13_shifted: int = (year - 1) if (month <= 2) else year + __inl126_result: Optional[int] = None + __inl124_quotient: int = __inl13_shifted // 400 + if (__inl13_shifted < 0) and ((__inl124_quotient * 400) != __inl13_shifted): + __inl126_result = __inl124_quotient - 1 + if __inl126_result is None: + __inl126_result = __inl124_quotient + __inl14_era: int = __inl126_result + __inl15_year_of_era: int = __inl13_shifted - (__inl14_era * 400) + __inl16_month_term: int = (month - 3) if (month > 2) else (month + 9) + __inl17_day_of_year: int = ((((153 * __inl16_month_term) + 2) // 5) + day) - 1 + __inl18_day_of_era: int = ( + ( + (__inl15_year_of_era * 365) + + ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := __inl15_year_of_era), + (__td_b := 4), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + - ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := __inl15_year_of_era), + (__td_b := 100), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + __inl17_day_of_year + __inl22_result: int = min( + max((((__inl14_era * 146097) + __inl18_day_of_era) - 719468), -719162), 2932896 + ) + days: int = __inl22_result + __inl23_shifted: int = days + 719468 + __inl129_result: Optional[int] = None + __inl127_quotient: int = __inl23_shifted // 146097 + if (__inl23_shifted < 0) and ((__inl127_quotient * 146097) != __inl23_shifted): + __inl129_result = __inl127_quotient - 1 + if __inl129_result is None: + __inl129_result = __inl127_quotient + __inl24_era: int = __inl129_result + __inl25_day_of_era: int = __inl23_shifted - (__inl24_era * 146097) + __inl26_year_of_era: int = ( + -(abs(__td_a) // abs(__td_b)) + if ( + ( + __td_a := ( + ( + ( + __inl25_day_of_era + - ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := __inl25_day_of_era), + (__td_b := 1460), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + + ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := __inl25_day_of_era), + (__td_b := 36524), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + - ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := __inl25_day_of_era), + (__td_b := 146096), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + ), + (__td_b := 365), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + __inl27_year: int = __inl26_year_of_era + (__inl24_era * 400) + __inl28_day_of_year: int = __inl25_day_of_era - ( + ( + (365 * __inl26_year_of_era) + + ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := __inl26_year_of_era), + (__td_b := 4), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + - ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := __inl26_year_of_era), + (__td_b := 100), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + __inl29_month_prime: int = ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := ((5 * __inl28_day_of_year) + 2)), + (__td_b := 153), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + __inl30_month: int = ( + (__inl29_month_prime + 3) + if (__inl29_month_prime < 10) + else (__inl29_month_prime - 9) + ) + __inl32_result: int = min( + max(((__inl27_year + 1) if (__inl30_month <= 2) else __inl27_year), 1), 9999 + ) + if ((__inl32_result != year) or (month_from_days(days) != month)) or ( + day_from_days(days) != day + ): + return None + return days + + +def year_from_days(days: int) -> int: + """The year of a date given as days since 1970-01-01.""" + shifted: int = days + 719468 + __inl3_result: Optional[int] = None + __inl1_quotient: int = shifted // 146097 + if (shifted < 0) and ((__inl1_quotient * 146097) != shifted): + __inl3_result = __inl1_quotient - 1 + if __inl3_result is None: + __inl3_result = __inl1_quotient + era: int = __inl3_result + day_of_era: int = shifted - (era * 146097) + year_of_era: int = ( + -(abs(__td_a) // abs(__td_b)) + if ( + ( + __td_a := ( + ( + ( + day_of_era + - ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := day_of_era), + (__td_b := 1460), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + + ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := day_of_era), + (__td_b := 36524), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + - ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := day_of_era), + (__td_b := 146096), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + ), + (__td_b := 365), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + year: int = year_of_era + (era * 400) + day_of_year: int = day_of_era - ( + ( + (365 * year_of_era) + + ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := year_of_era), + (__td_b := 4), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + ) + - ( + -(abs(__td_a) // abs(__td_b)) + if ((__td_a := year_of_era), (__td_b := 100), (__td_a < 0) != (__td_b < 0))[ + 2 + ] + else abs(__td_a) // abs(__td_b) + ) + ) + month_prime: int = ( + -(abs(__td_a) // abs(__td_b)) + if ( + (__td_a := ((5 * day_of_year) + 2)), + (__td_b := 153), + (__td_a < 0) != (__td_b < 0), + )[2] + else abs(__td_a) // abs(__td_b) + ) + month: int = (month_prime + 3) if (month_prime < 10) else (month_prime - 9) + return min(max(((year + 1) if (month <= 2) else year), 1), 9999) diff --git a/core/out/rust/API.json b/core/out/rust/API.json new file mode 100644 index 000000000..4408a0b6f --- /dev/null +++ b/core/out/rust/API.json @@ -0,0 +1,252 @@ +{ + "functions": [ + { + "name": "format_cnpj", + "module": "src/format_cnpj.rs", + "params": [ + { + "name": "value", + "type": "String" + }, + { + "name": "options", + "type": "FormatCnpjOptions" + } + ], + "returns": "String", + "effects": [] + }, + { + "name": "format_currency", + "module": "src/format_currency.rs", + "params": [ + { + "name": "value", + "type": "i64" + }, + { + "name": "symbol", + "type": "bool" + } + ], + "returns": "String", + "effects": [] + }, + { + "name": "generate_cnpj", + "module": "src/generate_cnpj.rs", + "params": [ + { + "name": "env", + "type": "&dyn Capabilities" + } + ], + "returns": "String", + "effects": [ + "env" + ] + }, + { + "name": "generate_cpf", + "module": "src/generate_cpf.rs", + "params": [ + { + "name": "env", + "type": "&dyn Capabilities" + } + ], + "returns": "String", + "effects": [ + "env" + ] + }, + { + "name": "get_address_info_by_cep", + "module": "src/get_address_info_by_cep.rs", + "params": [ + { + "name": "cep", + "type": "String" + }, + { + "name": "env", + "type": "&dyn Capabilities" + } + ], + "returns": "AddressInfo", + "effects": [ + "Fail", + "Fail", + "env" + ] + }, + { + "name": "get_holidays", + "module": "src/get_holidays.rs", + "params": [ + { + "name": "year", + "type": "i64" + } + ], + "returns": "Vec", + "effects": [] + }, + { + "name": "is_business_day", + "module": "src/is_business_day.rs", + "params": [ + { + "name": "value", + "type": "i64" + }, + { + "name": "include_optional", + "type": "bool" + } + ], + "returns": "bool", + "effects": [] + }, + { + "name": "is_valid_cnpj", + "module": "src/is_valid_cnpj.rs", + "params": [ + { + "name": "cnpj", + "type": "String" + }, + { + "name": "version", + "type": "String" + } + ], + "returns": "bool", + "effects": [] + }, + { + "name": "is_valid_cpf", + "module": "src/is_valid_cpf.rs", + "params": [ + { + "name": "cpf", + "type": "String" + } + ], + "returns": "bool", + "effects": [] + } + ], + "seams": [ + { + "name": "generate_cnpj", + "publicName": "generate_cnpj", + "module": "src/generate_cnpj.rs", + "params": [ + { + "name": "env", + "type": "&dyn Capabilities" + } + ], + "returns": "String", + "hasWrapper": false + }, + { + "name": "generate_cpf", + "publicName": "generate_cpf", + "module": "src/generate_cpf.rs", + "params": [ + { + "name": "env", + "type": "&dyn Capabilities" + } + ], + "returns": "String", + "hasWrapper": false + }, + { + "name": "get_address_info_by_cep", + "publicName": "get_address_info_by_cep", + "module": "src/get_address_info_by_cep.rs", + "params": [ + { + "name": "cep", + "type": "String" + }, + { + "name": "env", + "type": "&dyn Capabilities" + } + ], + "returns": "AddressInfo", + "hasWrapper": false + } + ], + "records": [ + { + "name": "FormatCnpjOptions", + "fields": [ + { + "name": "pad", + "type": "bool" + }, + { + "name": "version", + "type": "String" + }, + { + "name": "obfuscate", + "type": "bool" + } + ] + }, + { + "name": "AddressInfo", + "fields": [ + { + "name": "cep", + "type": "String" + }, + { + "name": "state", + "type": "String" + }, + { + "name": "city", + "type": "String" + }, + { + "name": "neighborhood", + "type": "String" + }, + { + "name": "street", + "type": "String" + } + ] + }, + { + "name": "Holiday", + "fields": [ + { + "name": "name", + "type": "String" + }, + { + "name": "date", + "type": "i64" + }, + { + "name": "r#type", + "type": "String" + } + ] + } + ], + "errors": [ + "GetAddressInfoByCepError", + "GetAddressInfoByCepNotFoundError", + "GetAddressInfoByCepValidationError", + "HttpError" + ] +} diff --git a/core/out/rust/Cargo.lock b/core/out/rust/Cargo.lock new file mode 100644 index 000000000..49f8cd58a --- /dev/null +++ b/core/out/rust/Cargo.lock @@ -0,0 +1,7 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "coreout" +version = "0.1.0" diff --git a/core/out/rust/Cargo.toml b/core/out/rust/Cargo.toml new file mode 100644 index 000000000..59eb0ba1c --- /dev/null +++ b/core/out/rust/Cargo.toml @@ -0,0 +1,10 @@ +[package] +name = "coreout" +version = "0.1.0" +edition = "2021" + +[[bin]] +name = "driver" +path = "src/bin/driver.rs" + +[dependencies] diff --git a/core/out/rust/LOWERING.md b/core/out/rust/LOWERING.md new file mode 100644 index 000000000..6b8cc3061 --- /dev/null +++ b/core/out/rust/LOWERING.md @@ -0,0 +1,310 @@ +# Lowering selections — rust + +Generated by the engine. Each row is one operation, the argument types it was called +with, the implementation that was selected, and the rule that decided it. + +| operation | argument types | implementation | why | +| --- | --- | --- | --- | +| `clock.sleep` | `Duration` | native | only candidate, cost none/constant | +| `core.eq` | `"1" \| "2", "2"` | native | only candidate, cost none/constant | +| `core.eq` | `"national" \| "optional" \| "religious" \| "state", "optional"` | native | only candidate, cost none/constant | +| `core.eq` | `Ascii[1], Ascii[1]` | native | only candidate, cost none/constant | +| `core.eq` | `Ascii[1], Digits[1]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[-1..1], Int[0..0]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[-2..2], Int[0..0]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[-48..79], Int[0..9]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[-1..-1]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[0..127]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[10..10]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[110..110]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[114..114]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[116..116]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[117..117]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[13..13]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[32..32]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[34..34]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[58..58]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[9..9]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[92..92]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..2147483647], Int[11..11]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..2147483647], Int[14..14]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..3506328], Int[306..-1]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..9], Int[0..9]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..9600], Int[0..-1]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..12], Int[1..12]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..31], Int[1..31]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..7], Int[6..6]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..7], Int[7..7]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..9999], Int[1..9999]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[48..57], Int[48..57]` | native | only candidate, cost none/constant | +| `core.eq` | `String[0..2147483647], Ascii[0]` | native | only candidate, cost none/constant | +| `date.addDays` | `CivilDate, Int[-2..-2]` | library | only candidate, cost none/constant | +| `date.addDays` | `CivilDate, Int[-47..-47]` | library | only candidate, cost none/constant | +| `date.addDays` | `CivilDate, Int[60..60]` | library | only candidate, cost none/constant | +| `date.clampEpochDays` | `Int[0..0]` | native | only candidate, cost none/constant | +| `date.compare` | `CivilDate, CivilDate` | native | only candidate, cost none/constant | +| `date.dayOfWeek` | `CivilDate` | native | only candidate, cost none/constant | +| `date.fromYmd` | `Int[1900..2099], Int[1..1], Int[1..1]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[10..10], Int[12..12]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[11..11], Int[15..15]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[11..11], Int[2..2]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[12..12], Int[25..25]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[3..3], Int[22..31]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[4..4], Int[1..25]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[4..4], Int[21..21]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[5..5], Int[1..1]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[1900..2099], Int[9..9], Int[7..7]` | portable | only candidate, cost none/linear | +| `date.fromYmd` | `Int[2024..2099], Int[11..11], Int[20..20]` | portable | only candidate, cost none/linear | +| `date.year` | `CivilDate` | portable | only candidate, cost none/constant | +| `dec.abs` | `Decimal<2>` | native | only candidate, cost none/constant | +| `dec.isNegative` | `Decimal<2>` | native | only candidate, cost none/constant | +| `dec.unscaled` | `Decimal<2>` | native | only candidate, cost none/constant | +| `http.request` | `HttpRequest` | native | only candidate, cost many/linear | +| `int.add` | `Int[-10012..20013], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-146097..3506328], Int[-3506503..3798697]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-14618796..14618800], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-238871..9], Int[3..3]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-3504000..3795635], Int[-2400..2599]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-3506503..3798330], Int[0..367]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-3508380..3800745], Int[-2403..2603]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-3508623..3800862], Int[-95..103]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-36547263..36546651], Int[2..2]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-36547330..36546740], Int[2..2]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-7..35], Int[114..114]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-719162..2932896], Int[719468..719468]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-9612..10413], Int[-400..9600]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..1683], Int[2..2]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..17], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..18], Int[0..319]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..2147483646], Int[0..4]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..2147483646], Int[5..5]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..29], Int[0..6]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..30], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..337], Int[0..132]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..337], Int[1..31]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..342], Int[19..20]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..65520], Int[0..15]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..720], Int[0..90]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..891], Int[0..81]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..891], Int[0..99]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..9007199254740991], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..9007199254740991], Int[2..2]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..9007199254740991], Int[6..6]` | native | only candidate, cost none/constant | +| `int.add` | `Int[1..2], Int[9..9]` | native | only candidate, cost none/constant | +| `int.add` | `Int[1..31], Int[0..31]` | native | only candidate, cost none/constant | +| `int.add` | `Int[32..32], Int[0..6]` | native | only candidate, cost none/constant | +| `int.add` | `Int[32..38], Int[0..48]` | native | only candidate, cost none/constant | +| `int.add` | `Int[5..2147483658], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[5..2147483659], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[6..2147483667], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[6..2147483668], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[8..352], Int[15..15]` | native | only candidate, cost none/constant | +| `int.add` | `Int[9..2147483671], Int[0..3]` | native | only candidate, cost none/constant | +| `int.div` | `Int[-3506022..3798461], Int[1460..1460]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[-3506022..3798461], Int[146096..146096]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[-3506022..3798461], Int[36524..36524]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[-3508743..3800988], Int[365..365]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[-36547261..36546653], Int[5..5]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[-36547328..36546742], Int[153..153]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[-9600..10399], Int[100..100]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[-9600..10399], Int[4..4]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[-9612..10413], Int[100..100]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[-9612..10413], Int[4..4]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[0..469], Int[451..451]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[0..99], Int[4..4]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[0..9999], Int[400..400]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[107..149], Int[31..31]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[19..20], Int[4..4]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[1900..2099], Int[100..100]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[2..1685], Int[5..5]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.div` | `Int[306..3652364], Int[146097..146097]` | native | only candidate, cost none/constant; Rust's `/` truncates toward zero, which is the Core's rule | +| `int.ge` | `Int[0..1114111], Int[0..0]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..1114111], Int[48..48]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..1114111], Int[65..65]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..1114111], Int[97..97]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..127], Int[65..65]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..17], Int[0..2147483647]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..599], Int[200..200]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[1900..2099], Int[2024..2024]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[0..14], Int[0..0]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[0..2], Int[0..0]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1..12], Int[2..2]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1..9007199254740991], Int[12..12]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1..9007199254740991], Int[31..31]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1..9007199254740991], Int[9999..9999]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1900..9999], Int[2099..2099]` | native | only candidate, cost none/constant | +| `int.le` | `Int[-238868..238858], Int[2..2]` | native | only candidate, cost none/constant | +| `int.le` | `Int[1..12], Int[2..2]` | native | only candidate, cost none/constant | +| `int.le` | `Int[22..56], Int[31..31]` | native | only candidate, cost none/constant | +| `int.le` | `Int[48..1114111], Int[57..57]` | native | only candidate, cost none/constant | +| `int.le` | `Int[65..1114111], Int[70..70]` | native | only candidate, cost none/constant | +| `int.le` | `Int[65..127], Int[90..90]` | native | only candidate, cost none/constant | +| `int.le` | `Int[97..1114111], Int[102..102]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[-238871..238867], Int[10..10]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[-9007199254740991..9007199254740991], Int[1..1]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[0..10], Int[2..2]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[0..17], Int[0..2147483647]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[0..4294967295], Int[4294967287..4294967296]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[0..9999], Int[0..0]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[1..9999], Int[1900..1900]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[200..599], Int[300..300]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[306..3652364], Int[0..0]` | native | only candidate, cost none/constant | +| `int.max` | `Int[-10012..20014], Int[1..1]` | native | only candidate, cost none/constant | +| `int.max` | `Int[-14618795..14618801], Int[1..1]` | native | only candidate, cost none/constant | +| `int.max` | `Int[-238868..238858], Int[1..1]` | native | only candidate, cost none/constant | +| `int.max` | `Int[-4372068..6585557], Int[-719162..-719162]` | native | only candidate, cost none/constant | +| `int.max` | `Int[1..15], Int[0..0]` | native | only candidate, cost none/constant | +| `int.max` | `Int[1..62], Int[22..22]` | native | only candidate, cost none/constant | +| `int.min` | `Int[-719162..6585557], Int[2932896..2932896]` | native | only candidate, cost none/constant | +| `int.min` | `Int[0..65535], Int[65535..65535]` | native | only candidate, cost none/constant | +| `int.min` | `Int[1..14618801], Int[31..31]` | native | only candidate, cost none/constant | +| `int.min` | `Int[1..20014], Int[9999..9999]` | native | only candidate, cost none/constant | +| `int.min` | `Int[1..238858], Int[12..12]` | native | only candidate, cost none/constant | +| `int.min` | `Int[22..62], Int[56..56]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[-14..14], Int[3..3]` | native | only candidate, cost none/constant; Rust's `%` takes the sign of the dividend, which is the Core's rule | +| `int.mod` | `Int[0..4294967295], Int[10..10]` | native | only candidate, cost none/constant; Rust's `%` takes the sign of the dividend, which is the Core's rule | +| `int.mod` | `Int[0..810], Int[11..11]` | native | only candidate, cost none/constant; Rust's `%` takes the sign of the dividend, which is the Core's rule | +| `int.mod` | `Int[0..86], Int[7..7]` | native | only candidate, cost none/constant; Rust's `%` takes the sign of the dividend, which is the Core's rule | +| `int.mod` | `Int[0..972], Int[11..11]` | native | only candidate, cost none/constant; Rust's `%` takes the sign of the dividend, which is the Core's rule | +| `int.mod` | `Int[0..99], Int[4..4]` | native | only candidate, cost none/constant; Rust's `%` takes the sign of the dividend, which is the Core's rule | +| `int.mod` | `Int[0..990], Int[11..11]` | native | only candidate, cost none/constant; Rust's `%` takes the sign of the dividend, which is the Core's rule | +| `int.mod` | `Int[107..149], Int[31..31]` | native | only candidate, cost none/constant; Rust's `%` takes the sign of the dividend, which is the Core's rule | +| `int.mod` | `Int[19..20], Int[4..4]` | native | only candidate, cost none/constant; Rust's `%` takes the sign of the dividend, which is the Core's rule | +| `int.mod` | `Int[1900..2099], Int[100..100]` | native | only candidate, cost none/constant; Rust's `%` takes the sign of the dividend, which is the Core's rule | +| `int.mod` | `Int[1900..2099], Int[19..19]` | native | only candidate, cost none/constant; Rust's `%` takes the sign of the dividend, which is the Core's rule | +| `int.mod` | `Int[23..367], Int[30..30]` | native | only candidate, cost none/constant; Rust's `%` takes the sign of the dividend, which is the Core's rule | +| `int.mul` | `Int[-1..24], Int[146097..146097]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[-1..24], Int[400..400]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[-9600..10399], Int[365..365]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..1], Int[31..31]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..24], Int[146097..146097]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..24], Int[400..400]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..4095], Int[16..16]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..9], Int[2..10]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..9], Int[2..11]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..9], Int[2..9]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[11..11], Int[0..29]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[153..153], Int[-238871..238867]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[153..153], Int[0..11]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[19..19], Int[0..18]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[2..2], Int[0..24]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[2..2], Int[0..3]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[22..22], Int[0..6]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[365..365], Int[-9612..10413]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[5..5], Int[-7309466..7309348]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[7..7], Int[0..1]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3506022..3798461], Int[-2401..2601]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3506022..3798461], Int[-3510887..3803444]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3506400..3798234], Int[-96..103]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3508718..3800965], Int[-23..25]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3510783..3803348], Int[-96..104]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3652600..7305025], Int[719468..719468]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-7309466..7309348], Int[-7309452..7309330]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..127], Int[48..48]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..15], Int[1..14]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..24], Int[1..1]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..35], Int[0..7]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..9999], Int[-400..9600]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[1..368], Int[1..1]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[1..9999], Int[1..1]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[10..10], Int[0..8]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[10..238867], Int[9..9]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[11..11], Int[0..9]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[11..11], Int[2..10]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[14..358], Int[6..6]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[19..362], Int[4..5]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[3..12], Int[3..3]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[3..17], Int[2..2]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[3..4], Int[3..3]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[3..86], Int[0..3]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[306..3652364], Int[-146097..3506328]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[32..56], Int[31..31]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[32..86], Int[0..29]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[48..57], Int[48..48]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[65..70], Int[55..55]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[97..102], Int[87..87]` | native | only candidate, cost none/constant | +| `opt.isNone` | `Option` | native | only candidate, cost none/constant | +| `opt.isNone` | `Option` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Ascii[0]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, CivilDate` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[-1..-1]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[0..0]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[48..48]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[-2..-2]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[0..0]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[48..48]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Ascii[0]` | native | only candidate, cost none/constant | +| `opt.unwrap` | `Option` | native | only candidate, cost none/constant | +| `opt.unwrap` | `Option` | native | only candidate, cost none/constant | +| `random.nextU32` | `` | native | only candidate, cost none/constant | +| `re.retain` | `String[0..2147483647]` | native | only candidate, cost one/linear; a byte-wise pass, not `.chars()`, when every retained range is ASCII (every class this project uses is: digits, letters) -- `.chars()` decodes the whole input as UTF-8 scalars before the filter ever runs, which was measured as the largest cost in `isValidCpf` once the regex engine itself stopped being one (`engine/docs/progress.md` §8); a byte never needs decoding to be range-tested, and a multi-byte scalar's bytes are all >= 0x80, so every one of them fails an ASCII range test on its own and is dropped exactly as it would be by testing the decoded scalar -- a non-ASCII input keeps working, just without ever paying to decode it. The ASCII branch writes straight into the `String` it returns with a plain `for` loop over `.bytes()`, rather than `.bytes().filter(...).collect::>()` followed by `String::from_utf8(..).unwrap()`: the iterator-adaptor chain and the `Vec` it builds only to hand to a UTF-8 validator that then has to re-walk it are both pure overhead here, since every byte the loop pushes is already a range-tested ASCII byte and therefore already valid UTF-8 on its own -- measured at roughly half the cost of the iterator-chain form on an 11-byte input (`is_valid_cpf`'s own `keep_digits` call), with no `unsafe` needed to get there. | +| `re.test` | `String[0..2147483647]` | library | only candidate, cost none/linear; a dedicated straight-line scanner (no allocation, one pass) when the pattern is a chain of character-class runs with no alternation and no two adjacent variable-length runs that could overlap; otherwise a backtracking matcher over a static pattern tree, also allocation-free — see engine/src/targets/rust/index.ts's "Regex" section for the rule | +| `seq.at` | `List[0..2147483647], Int[0..2147483650]` | library | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[0..9007199254740991]` | library | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[1..9007199254740992]` | library | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[5..2147483658]` | library | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[5..2147483659]` | library | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[6..2147483667]` | library | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[6..2147483668]` | library | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[9..2147483674]` | library | only candidate, cost none/constant | +| `seq.at` | `List[5..5], Int[0..4]` | library | only candidate, cost none/constant | +| `seq.at` | `List[0..15], Int[0..14]` | library | only candidate, cost none/constant | +| `seq.get` | `List[12..12], Int[0..11]` | library | only candidate, cost none/constant | +| `seq.len` | `List[0..2147483647]` | native | only candidate, cost none/constant | +| `seq.len` | `List[5..5]` | native | only candidate, cost none/constant | +| `seq.len` | `List[12..12]` | native | only candidate, cost none/constant | +| `seq.len` | `List[0..15]` | native | only candidate, cost none/constant | +| `seq.push` | `` | native | only candidate, cost none/constant | +| `seq.sortStableBy` | `List[12..13], (Holiday) => CivilDate` | native | only candidate, cost one/nlogn | +| `str.asciiUpper` | `Ascii[0..2147483647]` | native | only candidate, cost one/linear; `to_ascii_uppercase` is only ASCII-equivalent on ASCII input | +| `str.asciiUpper` | `String[0..2147483647]` | native | native, cost one/linear; mapping only a-z, leaving every other scalar alone, is the Core's rule for any input; rejected `to_ascii_uppercase` is only ASCII-equivalent on ASCII input | +| `str.charAtOpt` | `Ascii[0..2147483647], Int[0..17]` | library | only candidate, cost one/linear | +| `str.charAtOpt` | `Ascii[18], Int[0..17]` | library | only candidate, cost one/linear | +| `str.codeAt` | `Ascii[14], Int[12..12]` | native | only candidate, cost none/constant; indexing a byte string yields a byte | +| `str.codeAt` | `Ascii[14], Int[13..13]` | native | only candidate, cost none/constant; indexing a byte string yields a byte | +| `str.codeAt` | `Digits[11], Int[0..0]` | native | only candidate, cost none/constant; indexing a byte string yields a byte | +| `str.codeAt` | `Digits[11], Int[0..8]` | native | only candidate, cost none/constant; indexing a byte string yields a byte | +| `str.codeAt` | `Digits[11], Int[1..10]` | native | only candidate, cost none/constant; indexing a byte string yields a byte | +| `str.codeAt` | `Digits[11], Int[10..10]` | native | only candidate, cost none/constant; indexing a byte string yields a byte | +| `str.codeAt` | `Digits[11], Int[9..9]` | native | only candidate, cost none/constant; indexing a byte string yields a byte | +| `str.codeAt` | `Digits[12], Int[0..0]` | native | only candidate, cost none/constant; indexing a byte string yields a byte | +| `str.codeAt` | `Digits[12], Int[1..11]` | native | only candidate, cost none/constant; indexing a byte string yields a byte | +| `str.codeAt` | `Digits[14], Int[0..0]` | native | only candidate, cost none/constant; indexing a byte string yields a byte | +| `str.codeAt` | `Digits[14], Int[0..11]` | native | only candidate, cost none/constant; indexing a byte string yields a byte | +| `str.codeAt` | `Digits[14], Int[1..13]` | native | only candidate, cost none/constant; indexing a byte string yields a byte | +| `str.codeAtOpt` | `Ascii[0..2147483647], Int[0..2147483646]` | library | only candidate, cost none/constant | +| `str.codePoints` | `Ascii[5]` | native | only candidate, cost one/linear; an ASCII byte is already its own code point, so `.bytes()` needs no UTF-8 decode at all, unlike `code_points`' `.chars()` below | +| `str.codePoints` | `Digits[0..15]` | native | only candidate, cost one/linear; an ASCII byte is already its own code point, so `.bytes()` needs no UTF-8 decode at all, unlike `code_points`' `.chars()` below | +| `str.codePoints` | `String[0..2147483647]` | library | library, cost one/linear; rejected an ASCII byte is already its own code point, so `.bytes()` needs no UTF-8 decode at all, unlike `code_points`' `.chars()` below | +| `str.concat` | `Ascii[0..17], Ascii[1]` | library | only candidate, cost one/linear; `concat2` borrows both operands, unlike `+`, which would consume the left one | +| `str.concat` | `Ascii[0..3], Ascii[1..47]` | library | only candidate, cost one/linear; `concat2` borrows both operands, unlike `+`, which would consume the left one | +| `str.concat` | `Ascii[2], Ascii[1]` | library | only candidate, cost one/linear; `concat2` borrows both operands, unlike `+`, which would consume the left one | +| `str.concat` | `Ascii[36], Digits[8] matches ^[0-9]{8}$` | library | only candidate, cost one/linear; `concat2` borrows both operands, unlike `+`, which would consume the left one | +| `str.concat` | `Digits[12], Digits[2]` | library | only candidate, cost one/linear; `concat2` borrows both operands, unlike `+`, which would consume the left one | +| `str.concat` | `Digits[9], Digits[2]` | library | only candidate, cost one/linear; `concat2` borrows both operands, unlike `+`, which would consume the left one | +| `str.concatAll` | `Ascii[0..30], Ascii[1], Ascii[0..16]` | native | only candidate, cost one/linear; one buffer sized once, not a chain of reallocate-and-copy | +| `str.concatAll` | `Ascii[1], Ascii[0..3], Ascii[1..47]` | native | only candidate, cost one/linear; one buffer sized once, not a chain of reallocate-and-copy | +| `str.concatAll` | `Ascii[1], Ascii[3], Ascii[1]` | native | only candidate, cost one/linear; one buffer sized once, not a chain of reallocate-and-copy | +| `str.concatAll` | `Ascii[25], Digits[8] matches ^[0-9]{8}$, Ascii[6]` | native | only candidate, cost one/linear; one buffer sized once, not a chain of reallocate-and-copy | +| `str.concatAll` | `Digits[1], Digits[1], Digits[1], Digits[1], Digits[1], Digits[1], Digits[1], Digits[1], Digits[1], Digits[1], Digits[1], Digits[1]` | native | only candidate, cost one/linear; one buffer sized once, not a chain of reallocate-and-copy | +| `str.concatAll` | `Digits[1], Digits[1], Digits[1], Digits[1], Digits[1], Digits[1], Digits[1], Digits[1], Digits[1]` | native | only candidate, cost one/linear; one buffer sized once, not a chain of reallocate-and-copy | +| `str.concatAll` | `Digits[12], Digits[1], Digits[1]` | native | only candidate, cost one/linear; one buffer sized once, not a chain of reallocate-and-copy | +| `str.concatAll` | `Digits[9], Digits[1], Digits[1]` | native | only candidate, cost one/linear; one buffer sized once, not a chain of reallocate-and-copy | +| `str.fromCodePoints` | `List[0..2147483647]` | library | library, cost one/linear; rejected every code point this project ever builds this way is proven ASCII (`group_thousands`'s `out`), so its value is its whole UTF-8 encoding -- pushed straight into the `String` this builds, one byte-as-char per point, instead of `from_code_points`'s `char::from_u32` round trip below, which decodes a full scalar this project never produces | +| `str.fromCodePoints` | `List[0..30]` | native | only candidate, cost one/linear; every code point this project ever builds this way is proven ASCII (`group_thousands`'s `out`), so its value is its whole UTF-8 encoding -- pushed straight into the `String` this builds, one byte-as-char per point, instead of `from_code_points`'s `char::from_u32` round trip below, which decodes a full scalar this project never produces | +| `str.fromCodePoints` | `List[1..1]` | native | only candidate, cost one/linear; every code point this project ever builds this way is proven ASCII (`group_thousands`'s `out`), so its value is its whole UTF-8 encoding -- pushed straight into the `String` this builds, one byte-as-char per point, instead of `from_code_points`'s `char::from_u32` round trip below, which decodes a full scalar this project never produces | +| `str.fromInt` | `Int[-9007199254740991..9007199254740991]` | native | native, cost one/linear; rejected a value proven to be a single decimal digit (`randomDigit`'s own call, after specialization narrows `randomBelow`'s result at this call site) needs no general-purpose integer formatter -- one ASCII byte pushed directly is the whole job, where `i64::to_string` below computes a digit count, allocates a buffer sized for it, and writes back to front even for a single digit. A `call` to a small support function, not a `raw` fragment that prints its argument as text immediately: `hoistConstantTables` (`backend/lower.ts`) walks the *structured* Target AST for a list literal to lift into a module-level constant, and only runs after every candidate's own `emit`, so a `raw` fragment that has already flattened an argument's call tree into a source string -- which this candidate's argument sometimes is, e.g. `cnpj_check_digit`'s own weight-table argument in `generate_cnpj` -- would hide a list literal nested inside it from that pass, the same way it stayed hidden before this whole node was a `call` in disguise. `str.fromInt`'s own general candidate below keeps that same discipline (`method`, not `raw`), for the same reason. | +| `str.fromInt` | `Int[0..9]` | native | only candidate, cost one/linear; a value proven to be a single decimal digit (`randomDigit`'s own call, after specialization narrows `randomBelow`'s result at this call site) needs no general-purpose integer formatter -- one ASCII byte pushed directly is the whole job, where `i64::to_string` below computes a digit count, allocates a buffer sized for it, and writes back to front even for a single digit. A `call` to a small support function, not a `raw` fragment that prints its argument as text immediately: `hoistConstantTables` (`backend/lower.ts`) walks the *structured* Target AST for a list literal to lift into a module-level constant, and only runs after every candidate's own `emit`, so a `raw` fragment that has already flattened an argument's call tree into a source string -- which this candidate's argument sometimes is, e.g. `cnpj_check_digit`'s own weight-table argument in `generate_cnpj` -- would hide a list literal nested inside it from that pass, the same way it stayed hidden before this whole node was a `call` in disguise. `str.fromInt`'s own general candidate below keeps that same discipline (`method`, not `raw`), for the same reason. | +| `str.len` | `Ascii[0..2147483647]` | native | only candidate, cost none/constant; `str::len` counts bytes, which equals the scalar count only for ASCII | +| `str.len` | `Ascii[18]` | native | only candidate, cost none/constant; `str::len` counts bytes, which equals the scalar count only for ASCII | +| `str.len` | `Ascii[3..17]` | native | only candidate, cost none/constant; `str::len` counts bytes, which equals the scalar count only for ASCII | +| `str.len` | `Digits[0..2147483647]` | native | only candidate, cost none/constant; `str::len` counts bytes, which equals the scalar count only for ASCII | +| `str.len` | `Digits[12]` | native | only candidate, cost none/constant; `str::len` counts bytes, which equals the scalar count only for ASCII | +| `str.padStart` | `Ascii[0..2147483647], Int[0..18], Digits[1]` | native | only candidate, cost one/linear; when both the value and the pad string are proven ASCII, a scalar count is a byte count, so the length check and the padding loop need no `Vec` at all -- `pad_start` below builds one just to learn `value.len()` and to hand `extend` something to iterate, which was measured at roughly half of `format_currency`'s own cost (`pad_start`'s call on the whole-part digits) for a value this short | +| `str.padStart` | `Ascii[1..17], Int[3..3], Digits[1]` | native | only candidate, cost one/linear; when both the value and the pad string are proven ASCII, a scalar count is a byte count, so the length check and the padding loop need no `Vec` at all -- `pad_start` below builds one just to learn `value.len()` and to hand `extend` something to iterate, which was measured at roughly half of `format_currency`'s own cost (`pad_start`'s call on the whole-part digits) for a value this short | +| `str.slice` | `Ascii[3..17], Int[0..0], Int[1..15]` | native | only candidate, cost one/linear; slicing an ASCII string cuts at byte boundaries, which are scalar boundaries too | +| `str.slice` | `Ascii[3..17], Int[1..15], Int[3..17]` | native | only candidate, cost one/linear; slicing an ASCII string cuts at byte boundaries, which are scalar boundaries too | +| `str.trim` | `String[0..2147483647]` | native | only candidate, cost one/linear; `trim_matches` takes the cut set explicitly, so the 25 code points are exact | +| `task.race` | `List<() => Option>[2..2]` | library | only candidate, cost many/linear; `std::thread::scope` plus an `mpsc` channel: one thread per task, first `Some` wins | + +Mix: 262 native, 27 library, 12 portable. diff --git a/core/out/rust/SOURCEMAP.json b/core/out/rust/SOURCEMAP.json new file mode 100644 index 000000000..89ffe2484 --- /dev/null +++ b/core/out/rust/SOURCEMAP.json @@ -0,0 +1,287 @@ +{ + "src/lib_digits.rs#keep_alphanumeric": { + "module": "lib/digits", + "start": 575, + "end": 684 + }, + "src/lib_digits.rs#keep_digits": { + "module": "lib/digits", + "start": 397, + "end": 481 + }, + "src/lib_digits.rs#is_repeated_run": { + "module": "lib/digits", + "start": 1135, + "end": 1358 + }, + "src/lib_digits.rs#digit_at": { + "module": "lib/digits", + "start": 738, + "end": 842 + }, + "src/lib_digits.rs#digit_at_2": { + "module": "lib/digits", + "start": 738, + "end": 842 + }, + "src/lib_digits.rs#digit_at_3": { + "module": "lib/digits", + "start": 738, + "end": 842 + }, + "src/lib_format.rs#pattern_slots": { + "module": "lib/format", + "start": 1345, + "end": 2086 + }, + "src/lib_format.rs#format_with_pattern": { + "module": "lib/format", + "start": 2182, + "end": 2821 + }, + "src/lib_format.rs#group_thousands": { + "module": "lib/format", + "start": 664, + "end": 1279 + }, + "src/format_cnpj.rs#format_cnpj": { + "module": "format-cnpj", + "start": 967, + "end": 1233 + }, + "src/format_currency.rs#format_currency": { + "module": "format-currency", + "start": 966, + "end": 1499 + }, + "src/lib_random.rs#random_below": { + "module": "lib/random", + "start": 1137, + "end": 1640 + }, + "src/lib_random.rs#random_digit": { + "module": "lib/random", + "start": 1680, + "end": 1747 + }, + "src/lib_cnpj.rs#random_cnpj_base": { + "module": "lib/cnpj", + "start": 2700, + "end": 2947 + }, + "src/lib_cnpj.rs#cnpj_check_digit": { + "module": "lib/cnpj", + "start": 687, + "end": 1184 + }, + "src/lib_cnpj.rs#has_letter": { + "module": "lib/cnpj", + "start": 1706, + "end": 2363 + }, + "src/lib_cnpj.rs#has_valid_cnpj_checksum": { + "module": "lib/cnpj", + "start": 1265, + "end": 1478 + }, + "src/lib_cnpj.rs#is_repeated_cnpj": { + "module": "lib/cnpj", + "start": 3028, + "end": 3247 + }, + "src/generate_cnpj.rs#generate_cnpj": { + "module": "generate-cnpj", + "start": 982, + "end": 1571 + }, + "src/lib_cpf.rs#random_cpf_base": { + "module": "lib/cpf", + "start": 999, + "end": 1196 + }, + "src/lib_cpf.rs#cpf_check_digit": { + "module": "lib/cpf", + "start": 394, + "end": 687 + }, + "src/lib_cpf.rs#cpf_check_digit_1": { + "module": "lib/cpf", + "start": 394, + "end": 687 + }, + "src/lib_cpf.rs#is_repeated": { + "module": "lib/cpf", + "start": 1283, + "end": 1499 + }, + "src/generate_cpf.rs#generate_cpf": { + "module": "generate-cpf", + "start": 1014, + "end": 1566 + }, + "src/get_address_info_by_cep.rs#get_with_retry": { + "module": "get-address-info-by-cep", + "start": 1086, + "end": 1491 + }, + "src/get_address_info_by_cep.rs#is_ok": { + "module": "get-address-info-by-cep", + "start": 1529, + "end": 1622 + }, + "src/get_address_info_by_cep.rs#fetch_via_cep": { + "module": "get-address-info-by-cep", + "start": 1701, + "end": 2300 + }, + "src/get_address_info_by_cep.rs#fetch_brasil_api": { + "module": "get-address-info-by-cep", + "start": 2351, + "end": 2957 + }, + "src/get_address_info_by_cep.rs#get_address_info_by_cep": { + "module": "get-address-info-by-cep", + "start": 3382, + "end": 3789 + }, + "src/lib_json.rs#matches_at": { + "module": "lib/json", + "start": 582, + "end": 1320 + }, + "src/lib_json.rs#is_space": { + "module": "lib/json", + "start": 1370, + "end": 1497 + }, + "src/lib_json.rs#hex_value": { + "module": "lib/json", + "start": 1568, + "end": 2072 + }, + "src/lib_json.rs#json_string_field": { + "module": "lib/json", + "start": 2300, + "end": 4239 + }, + "src/lib_civil.rs#civil_date": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "src/lib_civil.rs#civil_date_1": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "src/lib_civil.rs#civil_date_2": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "src/lib_civil.rs#civil_date_3": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "src/lib_civil.rs#civil_date_4": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "src/lib_civil.rs#civil_date_5": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "src/lib_civil.rs#civil_date_6": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "src/lib_civil.rs#civil_date_7": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "src/lib_civil.rs#civil_date_8": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "src/lib_civil.rs#civil_date_9": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "src/lib_civil.rs#civil_date_10": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "src/lib_easter.rs#easter_day_of_march": { + "module": "lib/easter", + "start": 527, + "end": 1128 + }, + "src/lib_easter.rs#easter_sunday": { + "module": "lib/easter", + "start": 1186, + "end": 1392 + }, + "src/get_holidays.rs#get_holidays": { + "module": "get-holidays", + "start": 898, + "end": 2426 + }, + "src/is_business_day.rs#is_business_day": { + "module": "is-business-day", + "start": 578, + "end": 1064 + }, + "src/is_valid_cnpj.rs#is_valid_cnpj": { + "module": "is-valid-cnpj", + "start": 1397, + "end": 2445 + }, + "src/is_valid_cpf.rs#is_valid_cpf": { + "module": "is-valid-cpf", + "start": 793, + "end": 1143 + }, + "src/std_date.rs#floor_div_1": { + "module": "std/date", + "start": 432, + "end": 655 + }, + "src/std_date.rs#days_from_civil": { + "module": "std/date", + "start": 752, + "end": 1560 + }, + "src/std_date.rs#floor_div": { + "module": "std/date", + "start": 432, + "end": 655 + }, + "src/std_date.rs#year_from_days": { + "module": "std/date", + "start": 1627, + "end": 2212 + }, + "src/std_date.rs#month_from_days": { + "module": "std/date", + "start": 2280, + "end": 2780 + }, + "src/std_date.rs#day_from_days": { + "module": "std/date", + "start": 2855, + "end": 3346 + }, + "src/std_date.rs#ymd_to_days": { + "module": "std/date", + "start": 3922, + "end": 4284 + } +} diff --git a/core/out/rust/fixtures.json b/core/out/rust/fixtures.json new file mode 100644 index 000000000..f05926010 --- /dev/null +++ b/core/out/rust/fixtures.json @@ -0,0 +1,27 @@ +{ + "https://viacep.com.br/ws/01310100/json/": { + "status": 200, + "body": "{\"cep\":\"01310-100\",\"logradouro\":\"Avenida Paulista\",\"bairro\":\"Bela Vista\",\"localidade\":\"São Paulo\",\"uf\":\"SP\"}", + "latencyMillis": 60 + }, + "https://brasilapi.com.br/api/cep/v1/01310100": { + "status": 200, + "body": "{\"cep\":\"01310100\",\"state\":\"SP\",\"city\":\"São Paulo\",\"neighborhood\":\"Bela Vista\",\"street\":\"Avenida Paulista\"}", + "latencyMillis": 20 + }, + "https://brasilapi.com.br/api/cep/v1/30130010": { + "status": 200, + "body": "{\"cep\":\"30130010\",\"state\":\"MG\",\"city\":\"Belo Horizonte\",\"neighborhood\":\"Centro\",\"street\":\"Avenida Afonso Pena\"}", + "latencyMillis": 40 + }, + "https://viacep.com.br/ws/99999999/json/": { + "status": 200, + "body": "{\"erro\":true}", + "latencyMillis": 10 + }, + "https://brasilapi.com.br/api/cep/v1/99999999": { + "status": 404, + "body": "{\"message\":\"not found\"}", + "latencyMillis": 10 + } +} diff --git a/core/out/rust/src/bin/driver.rs b/core/out/rust/src/bin/driver.rs new file mode 100644 index 000000000..222204295 --- /dev/null +++ b/core/out/rust/src/bin/driver.rs @@ -0,0 +1,272 @@ +// Code generated by the logic engine. DO NOT EDIT. +// source: _driver + +#![allow(clippy::needless_return, unused_parens, unused_imports)] + +use coreout::support::Capabilities; +use coreout::*; + +use coreout::json; + +// The variant name alone, the same family-membership test `errorName` gives the Go +// driver: `CoreError`'s derived `Debug` prints "VariantName { message: ... }", so the +// first token is it. +fn error_name(err: &coreout::errors::CoreError) -> String { + format!("{:?}", err) + .split_whitespace() + .next() + .unwrap_or("") + .to_string() +} + +/// The reference PCG32: same constants and default seed as the interpreter's, so a draw +/// matches the reference bit for bit. A fresh instance is built per case, the same way +/// the reference model starts a fresh interpreter -- and so a fresh generator -- per case. +struct Pcg32 { + state: u64, + increment: u64, +} + +impl Pcg32 { + fn new(seed: u64) -> Self { + let mut p = Pcg32 { + state: 0, + increment: 1442695040888963407, + }; + p.next(); + p.state = p.state.wrapping_add(seed); + p.next(); + p + } + + fn next(&mut self) -> i64 { + let previous = self.state; + self.state = previous + .wrapping_mul(6364136223846793005) + .wrapping_add(self.increment); + let xorshifted = (((previous >> 18) ^ previous) >> 27) as u32; + let rotation = (previous >> 59) as u32; + (xorshifted.rotate_right(rotation)) as i64 + } +} + +/// The interpreter's own default seed, used whenever Capabilities.seed is left unset. +const DEFAULT_SEED: u64 = 0x853c49e6748fea9b; + +struct Fixture { + status: i64, + body: String, + latency_millis: i64, +} + +/// The capability fake the differential harness drives: responses come from +/// fixtures.json, a missing URL is a transport error, and the scripted latency is what +/// decides a race. +struct FakeCapabilities { + fixtures: std::collections::HashMap, + random: std::sync::Mutex, +} + +impl Capabilities for FakeCapabilities { + fn request( + &self, + request: coreout::support::HttpRequest, + ) -> Option { + let fixture = self.fixtures.get(&request.url)?; + std::thread::sleep(std::time::Duration::from_millis( + fixture.latency_millis as u64, + )); + Some(coreout::support::HttpResponse { + status: fixture.status, + headers: vec![], + body: fixture.body.clone(), + }) + } + + fn now(&self) -> i64 { + 0 + } + + fn sleep(&self, millis: i64) { + std::thread::sleep(std::time::Duration::from_millis(millis as u64)); + } + + fn next_u32(&self) -> i64 { + self.random.lock().unwrap().next() + } +} + +fn load_fixtures() -> std::collections::HashMap { + let mut fixtures = std::collections::HashMap::new(); + if let Ok(raw) = std::fs::read_to_string("fixtures.json") { + if let json::Json::Object(entries) = json::parse(&raw) { + for (url, value) in entries { + fixtures.insert( + url, + Fixture { + status: value.get("status").map(|v| v.as_i64()).unwrap_or(0), + body: value + .get("body") + .map(|v| v.as_str().to_string()) + .unwrap_or_default(), + latency_millis: value.get("latencyMillis").map(|v| v.as_i64()).unwrap_or(0), + }, + ); + } + } + } + fixtures +} + +fn new_environment() -> Box { + Box::new(FakeCapabilities { + fixtures: load_fixtures(), + random: std::sync::Mutex::new(Pcg32::new(DEFAULT_SEED)), + }) +} + +fn dispatch(name: &str, args: &[json::Json], environment: &dyn Capabilities) -> json::Json { + match name { + "format-cnpj::formatCnpj" => { + let value = format_cnpj( + args[0].as_str(), + FormatCnpjOptions { + pad: args[1].get("pad").unwrap().as_bool(), + version: args[1].get("version").unwrap().as_str().to_string(), + obfuscate: args[1].get("obfuscate").unwrap().as_bool(), + }, + ); + coreout::json::Json::Object(vec![ + ("ok".to_string(), coreout::json::Json::Bool(true)), + ("value".to_string(), coreout::json::Json::String(value)), + ]) + } + "format-currency::formatCurrency" => { + let value = format_currency(args[0].as_i64(), args[1].as_bool()); + coreout::json::Json::Object(vec![ + ("ok".to_string(), coreout::json::Json::Bool(true)), + ("value".to_string(), coreout::json::Json::String(value)), + ]) + } + "generate-cnpj::generateCnpj" => { + let value = generate_cnpj(environment); + coreout::json::Json::Object(vec![ + ("ok".to_string(), coreout::json::Json::Bool(true)), + ("value".to_string(), coreout::json::Json::String(value)), + ]) + } + "generate-cpf::generateCpf" => { + let value = generate_cpf(environment); + coreout::json::Json::Object(vec![ + ("ok".to_string(), coreout::json::Json::Bool(true)), + ("value".to_string(), coreout::json::Json::String(value)), + ]) + } + "get-address-info-by-cep::getAddressInfoByCep" => { + match get_address_info_by_cep(args[0].as_str(), environment) { + Ok(value) => coreout::json::Json::Object(vec![ + ("ok".to_string(), coreout::json::Json::Bool(true)), + ("value".to_string(), { + let rec = value; + coreout::json::Json::Object(vec![ + ("cep".to_string(), coreout::json::Json::String(rec.cep)), + ("state".to_string(), coreout::json::Json::String(rec.state)), + ("city".to_string(), coreout::json::Json::String(rec.city)), + ( + "neighborhood".to_string(), + coreout::json::Json::String(rec.neighborhood), + ), + ( + "street".to_string(), + coreout::json::Json::String(rec.street), + ), + ]) + }), + ]), + Err(err) => coreout::json::Json::Object(vec![ + ("ok".to_string(), coreout::json::Json::Bool(false)), + ( + "error".to_string(), + coreout::json::Json::String(error_name(&err)), + ), + ]), + } + } + "get-holidays::getHolidays" => { + let value = get_holidays(args[0].as_i64()); + coreout::json::Json::Object(vec![ + ("ok".to_string(), coreout::json::Json::Bool(true)), + ( + "value".to_string(), + coreout::json::Json::Array( + value + .into_iter() + .map(|item| { + let rec = item; + coreout::json::Json::Object(vec![ + ("name".to_string(), coreout::json::Json::String(rec.name)), + ( + "date".to_string(), + coreout::json::Json::Number(rec.date as f64), + ), + ("type".to_string(), coreout::json::Json::String(rec.r#type)), + ]) + }) + .collect(), + ), + ), + ]) + } + "is-business-day::isBusinessDay" => { + let value = is_business_day(args[0].as_i64(), args[1].as_bool()); + coreout::json::Json::Object(vec![ + ("ok".to_string(), coreout::json::Json::Bool(true)), + ("value".to_string(), coreout::json::Json::Bool(value)), + ]) + } + "is-valid-cnpj::isValidCnpj" => { + let value = is_valid_cnpj(args[0].as_str(), args[1].as_str()); + coreout::json::Json::Object(vec![ + ("ok".to_string(), coreout::json::Json::Bool(true)), + ("value".to_string(), coreout::json::Json::Bool(value)), + ]) + } + "is-valid-cpf::isValidCpf" => { + let value = is_valid_cpf(args[0].as_str()); + coreout::json::Json::Object(vec![ + ("ok".to_string(), coreout::json::Json::Bool(true)), + ("value".to_string(), coreout::json::Json::Bool(value)), + ]) + } + _ => coreout::json::Json::Object(vec![ + ("ok".to_string(), coreout::json::Json::Bool(false)), + ( + "error".to_string(), + coreout::json::Json::String(format!("unknown function {}", name)), + ), + ]), + } +} + +fn main() { + use std::io::BufRead; + let stdin = std::io::stdin(); + for line in stdin.lock().lines() { + let line = line.unwrap(); + if line.trim().is_empty() { + continue; + } + let parsed = json::parse(&line); + let name = parsed + .get("fn") + .map(|v| v.as_str().to_string()) + .unwrap_or_default(); + let args: Vec = parsed + .get("args") + .map(|v| v.as_array().to_vec()) + .unwrap_or_default(); + let environment = new_environment(); + let result = dispatch(&name, &args, environment.as_ref()); + println!("{}", json::write(&result)); + } +} diff --git a/core/out/rust/src/errors.rs b/core/out/rust/src/errors.rs new file mode 100644 index 000000000..17eaab750 --- /dev/null +++ b/core/out/rust/src/errors.rs @@ -0,0 +1,29 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: errors + +/// Every domain error the project declares, as one flat enum: a nested fallible call is +/// `f(...)?` because every fallible function shares this one error type, with never a +/// mismatch for `?` to bridge. The sketch expected one type per utility; this is simpler, +/// and loses nothing `errors.Is`-shaped, since the variant itself is the family membership +/// test (`matches!(err, CoreError::SomeVariant { .. })`). +#[derive(Clone, Debug, PartialEq)] +pub enum CoreError { + HttpError { message: String }, + GetAddressInfoByCepError { message: String }, + GetAddressInfoByCepValidationError { message: String }, + GetAddressInfoByCepNotFoundError { message: String }, +} + +impl std::fmt::Display for CoreError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + CoreError::HttpError { message } => write!(f, "{}", message), + CoreError::GetAddressInfoByCepError { message } => write!(f, "{}", message), + CoreError::GetAddressInfoByCepValidationError { message } => write!(f, "{}", message), + CoreError::GetAddressInfoByCepNotFoundError { message } => write!(f, "{}", message), + } + } +} + +impl std::error::Error for CoreError {} diff --git a/core/out/rust/src/format_cnpj.rs b/core/out/rust/src/format_cnpj.rs new file mode 100644 index 000000000..a7b21e3a2 --- /dev/null +++ b/core/out/rust/src/format_cnpj.rs @@ -0,0 +1,34 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: format-cnpj +// content: 7541034a1f26 + +use crate::*; + +#[derive(Clone, Debug, PartialEq)] +pub struct FormatCnpjOptions { + pub pad: bool, + pub version: String, + pub obfuscate: bool, +} + +/// Formats a CNPJ value as `00.000.000/0000-00`. +/// +/// The core takes a string and a fully normalized options record; reading a number, a missing +/// options object or a truthy non-boolean is the DX's job. +pub fn format_cnpj(value: &str, options: FormatCnpjOptions) -> String { + let sanitized = (if (options.version == "2") { + keep_alphanumeric(value) + } else { + keep_digits(value) + }); + return format_with_pattern( + &sanitized, + &(if options.obfuscate { + "**.000.000/0000-**".to_string() + } else { + "00.000.000/0000-00".to_string() + }), + options.pad, + ); +} diff --git a/core/out/rust/src/format_currency.rs b/core/out/rust/src/format_currency.rs new file mode 100644 index 000000000..cff5f9f01 --- /dev/null +++ b/core/out/rust/src/format_currency.rs @@ -0,0 +1,54 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: format-currency +// content: 1e0947bdf328 + +use crate::*; + +/// Formats an exact amount in Brazilian Real, with two decimal places. +/// +/// The separators are the ones Lei nº 9.069/1995 art. 1º prescribes and the CLDR pt-BR data uses: +/// `.` between thousands, `,` before the centavos, and a non-breaking space after `R$`. A negative +/// amount puts the sign before the symbol, `-R$ 10,50`, the shape `Intl.NumberFormat` produces. +/// +/// Turning a host value into an exact amount is the DX's job, and so is the rounding that +/// conversion needs; see docs/contracts.md, which records exactly how the published package rounds. +pub fn format_currency(value: i64, symbol: bool) -> String { + let negative = (value < 0); + let unscaled = value.abs(); + let digits = crate::support::pad_start_ascii(&unscaled.to_string(), 3, "0"); + let cut = ((digits.len() as i64) - 2).max(0); + let whole = digits[0..cut as usize].to_string(); + let cents = digits[cut as usize..(digits.len() as i64) as usize].to_string(); + let body = { + let __piece0 = group_thousands(&keep_digits(&whole)); + let mut __buf = String::with_capacity(__piece0.len() + ",".len() + cents.len()); + __buf.push_str(&__piece0); + __buf.push(','); + __buf.push_str(¢s); + __buf + }; + let prefix = (if symbol { + crate::support::concat2("R$", &{ + let __pts = &[32]; + let mut __out = String::with_capacity(__pts.len()); + for &__p in __pts { + __out.push(__p as u8 as char); + } + __out + }) + } else { + "".to_string() + }); + return (if negative { + { + let mut __buf = String::with_capacity("-".len() + prefix.len() + body.len()); + __buf.push('-'); + __buf.push_str(&prefix); + __buf.push_str(&body); + __buf + } + } else { + crate::support::concat2(&prefix, &body) + }); +} diff --git a/core/out/rust/src/generate_cnpj.rs b/core/out/rust/src/generate_cnpj.rs new file mode 100644 index 000000000..0ed8df420 --- /dev/null +++ b/core/out/rust/src/generate_cnpj.rs @@ -0,0 +1,48 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: generate-cnpj +// content: 41b441a54f18 + +use crate::*; + +pub const GENERATE_CNPJ_TABLE1: &[i64] = &[5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]; + +pub const GENERATE_CNPJ_TABLE2: &[i64] = &[6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]; + +/// Generates a valid random CNPJ (Cadastro Nacional da Pessoa Jurídica) in the numeric format: 14 +/// digits, under the check digit rule both CNPJ versions share. +/// +/// Matches the published `generateCnpj()` called with no options: a random 8-digit root and +/// 4-digit branch (the "número de ordem"), redrawn while every digit of the 12-digit base is the +/// same, followed by its two check digits. The alphanumeric version and a chosen branch are DX +/// concerns layered on the same base and check digit rule, not a different generator. +pub fn generate_cnpj(env: &dyn Capabilities) -> String { + let mut base = random_cnpj_base(env); + for _attempt in 0..8 { + if !is_repeated_run(&base) { + break; + } + base = random_cnpj_base(env); + } + let first_digit = crate::support::digit_char(cnpj_check_digit( + &crate::support::concat2(&base, "00"), + GENERATE_CNPJ_TABLE1, + )); + let second_digit = crate::support::digit_char(cnpj_check_digit( + &{ + let mut __buf = String::with_capacity(base.len() + first_digit.len() + "0".len()); + __buf.push_str(&base); + __buf.push_str(&first_digit); + __buf.push('0'); + __buf + }, + GENERATE_CNPJ_TABLE2, + )); + return { + let mut __buf = String::with_capacity(base.len() + first_digit.len() + second_digit.len()); + __buf.push_str(&base); + __buf.push_str(&first_digit); + __buf.push_str(&second_digit); + __buf + }; +} diff --git a/core/out/rust/src/generate_cpf.rs b/core/out/rust/src/generate_cpf.rs new file mode 100644 index 000000000..5ef888e6d --- /dev/null +++ b/core/out/rust/src/generate_cpf.rs @@ -0,0 +1,40 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: generate-cpf +// content: 00fc978e8350 + +use crate::*; + +/// Generates a valid random CPF (Cadastro de Pessoas Físicas): 11 digits, under the check digit +/// rule (weights 10..2 and 11..2) the Receita Federal's Manual de Preenchimento da e-Financeira, +/// Anexo II specifies. +/// +/// Matches the published `generateCpf()` called with no state: a random 9-digit base — 8 digits +/// plus a região fiscal digit, also drawn at random here — redrawn while every digit of it is the +/// same, followed by its two check digits. The state code option is a DX concern: it only ever +/// picks which digit the 9th position draws from, never how the rest of the document is built. +pub fn generate_cpf(env: &dyn Capabilities) -> String { + let mut base = random_cpf_base(env); + for _attempt in 0..8 { + if !is_repeated_run(&base) { + break; + } + base = random_cpf_base(env); + } + let first_digit = + crate::support::digit_char(cpf_check_digit(&crate::support::concat2(&base, "00"))); + let second_digit = crate::support::digit_char(cpf_check_digit_1(&{ + let mut __buf = String::with_capacity(base.len() + first_digit.len() + "0".len()); + __buf.push_str(&base); + __buf.push_str(&first_digit); + __buf.push('0'); + __buf + })); + return { + let mut __buf = String::with_capacity(base.len() + first_digit.len() + second_digit.len()); + __buf.push_str(&base); + __buf.push_str(&first_digit); + __buf.push_str(&second_digit); + __buf + }; +} diff --git a/core/out/rust/src/get_address_info_by_cep.rs b/core/out/rust/src/get_address_info_by_cep.rs new file mode 100644 index 000000000..9d81acc67 --- /dev/null +++ b/core/out/rust/src/get_address_info_by_cep.rs @@ -0,0 +1,129 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: get-address-info-by-cep +// content: daa7ccc2e0b6 + +use crate::*; + +#[derive(Clone, Debug, PartialEq)] +pub struct AddressInfo { + pub cep: String, + pub state: String, + pub city: String, + pub neighborhood: String, + pub street: String, +} + +/// One GET, retried the way the published package retries: twice more, 250 ms apart. +fn get_with_retry(url: String, env: &dyn Capabilities) -> Option { + for attempt in 0..3 { + if (attempt > 0) { + env.sleep(250); + } + let response = env.request(crate::support::HttpRequest { + method: "GET".to_string(), + url: url.to_owned(), + headers: vec![], + body: "".to_string(), + timeout_millis: 10000, + }); + if response.is_some() { + return response.to_owned(); + } + } + return None; +} + +/// Whether the status is a 2xx. +fn is_ok(status: i64) -> bool { + return (200..300).contains(&status); +} + +/// ViaCEP answers a JSON object, and marks an unknown CEP with `"erro"`. +fn fetch_via_cep(cep: &str, env: &dyn Capabilities) -> Option { + let response = get_with_retry( + { + let mut __buf = String::with_capacity( + "https://viacep.com.br/ws/".len() + cep.len() + "/json/".len(), + ); + __buf.push_str("https://viacep.com.br/ws/"); + __buf.push_str(cep); + __buf.push_str("/json/"); + __buf + }, + env, + ); + if (response.is_none() || !is_ok(response.as_ref().unwrap().status.to_owned())) { + return None; + } + let code = json_string_field(&response.as_ref().unwrap().body, "cep").unwrap_or("".to_string()); + if code.is_empty() { + return None; + } + return Some(AddressInfo { + cep: keep_digits(&code), + state: json_string_field(&response.as_ref().unwrap().body, "uf").unwrap_or("".to_string()), + city: json_string_field(&response.as_ref().unwrap().body, "localidade") + .unwrap_or("".to_string()), + neighborhood: json_string_field(&response.as_ref().unwrap().body, "bairro") + .unwrap_or("".to_string()), + street: json_string_field(&response.as_ref().unwrap().body, "logradouro") + .unwrap_or("".to_string()), + }); +} + +/// BrasilAPI answers 404 for an unknown CEP. +fn fetch_brasil_api(cep: &str, env: &dyn Capabilities) -> Option { + let response = get_with_retry( + crate::support::concat2("https://brasilapi.com.br/api/cep/v1/", cep), + env, + ); + if (response.is_none() || !is_ok(response.as_ref().unwrap().status.to_owned())) { + return None; + } + let code = json_string_field(&response.as_ref().unwrap().body, "cep").unwrap_or("".to_string()); + if code.is_empty() { + return None; + } + return Some(AddressInfo { + cep: keep_digits(&code), + state: json_string_field(&response.as_ref().unwrap().body, "state") + .unwrap_or("".to_string()), + city: json_string_field(&response.as_ref().unwrap().body, "city").unwrap_or("".to_string()), + neighborhood: json_string_field(&response.as_ref().unwrap().body, "neighborhood") + .unwrap_or("".to_string()), + street: json_string_field(&response.as_ref().unwrap().body, "street") + .unwrap_or("".to_string()), + }); +} + +/// The address of a CEP, from the first service that answers. +/// +/// The two services are queried concurrently and the first answer wins; the losing request may +/// still finish, and its answer is dropped, which is why only idempotent GETs belong here. Each +/// request is retried twice, 250 ms apart, exactly as the published package does. Turning a host +/// value into the 8 digits this takes is the DX's job. +pub fn get_address_info_by_cep( + cep: &str, + env: &dyn Capabilities, +) -> Result { + if !(crate::support::re_match_0(cep)) { + return Err(CoreError::GetAddressInfoByCepValidationError { + message: "CEP inv\u{e1}lido".to_string(), + }); + } + let address = crate::support::race_first_some(vec![ + Box::new(|| { + return fetch_via_cep(cep, env); + }) as Box _ + Send>, + Box::new(|| { + return fetch_brasil_api(cep, env); + }) as Box _ + Send>, + ]); + if address.is_none() { + return Err(CoreError::GetAddressInfoByCepNotFoundError { + message: "CEP n\u{e3}o encontrado".to_string(), + }); + } + return Ok(address.unwrap()); +} diff --git a/core/out/rust/src/get_holidays.rs b/core/out/rust/src/get_holidays.rs new file mode 100644 index 000000000..d79497a9c --- /dev/null +++ b/core/out/rust/src/get_holidays.rs @@ -0,0 +1,93 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: get-holidays +// content: b4d687034433 + +use crate::*; + +#[derive(Clone, Debug, PartialEq)] +pub struct Holiday { + pub name: String, + pub date: i64, + pub r#type: String, +} + +/// The Brazilian national holidays of a year, sorted by date. +/// +/// The order is the one the published package produces: the fixed holidays in statutory order, +/// then the Easter-derived ones, sorted by date with a stable sort, so two holidays on the same +/// day keep the order they were built in. State holidays are not part of this pilot. +pub fn get_holidays(year: i64) -> Vec { + let mut holidays = vec![]; + holidays.push(Holiday { + name: "Ano novo".to_string(), + date: civil_date(year.to_owned()), + r#type: "national".to_string(), + }); + holidays.push(Holiday { + name: "Tiradentes".to_string(), + date: civil_date_1(year.to_owned()), + r#type: "national".to_string(), + }); + holidays.push(Holiday { + name: "Dia do trabalhador".to_string(), + date: civil_date_2(year.to_owned()), + r#type: "national".to_string(), + }); + holidays.push(Holiday { + name: "Independ\u{ea}ncia do Brasil".to_string(), + date: civil_date_3(year.to_owned()), + r#type: "national".to_string(), + }); + holidays.push(Holiday { + name: "Nossa Senhora Aparecida".to_string(), + date: civil_date_4(year.to_owned()), + r#type: "national".to_string(), + }); + holidays.push(Holiday { + name: "Finados".to_string(), + date: civil_date_5(year.to_owned()), + r#type: "national".to_string(), + }); + holidays.push(Holiday { + name: "Proclama\u{e7}\u{e3}o da Rep\u{fa}blica".to_string(), + date: civil_date_6(year.to_owned()), + r#type: "national".to_string(), + }); + holidays.push(Holiday { + name: "Natal".to_string(), + date: civil_date_7(year.to_owned()), + r#type: "national".to_string(), + }); + if (year >= 2024) { + holidays.push(Holiday { + name: "Dia da Consci\u{ea}ncia Negra".to_string(), + date: civil_date_8(year.to_owned()), + r#type: "national".to_string(), + }); + } + let easter = easter_sunday(year); + holidays.push(Holiday { + name: "Carnaval (ter\u{e7}a-feira)".to_string(), + date: crate::support::date_from_epoch_days(easter + -47).unwrap_or(easter), + r#type: "optional".to_string(), + }); + holidays.push(Holiday { + name: "Sexta-feira Santa".to_string(), + date: crate::support::date_from_epoch_days(easter + -2).unwrap_or(easter), + r#type: "national".to_string(), + }); + holidays.push(Holiday { + name: "P\u{e1}scoa".to_string(), + date: easter.to_owned(), + r#type: "religious".to_string(), + }); + holidays.push(Holiday { + name: "Corpus Christi".to_string(), + date: crate::support::date_from_epoch_days(easter + 60).unwrap_or(easter), + r#type: "optional".to_string(), + }); + return crate::support::sorted_stable_by(&holidays, |holiday: &Holiday| { + return holiday.date; + }); +} diff --git a/core/out/rust/src/is_business_day.rs b/core/out/rust/src/is_business_day.rs new file mode 100644 index 000000000..326d86064 --- /dev/null +++ b/core/out/rust/src/is_business_day.rs @@ -0,0 +1,34 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: is-business-day +// content: 3605fae387a8 + +use crate::*; + +/// Whether a date is a Brazilian business day (dia útil). +/// +/// A day is not a business day when it falls on a weekend, or when it is one of the holidays +/// `getHolidays` lists for its year. `includeOptional` decides whether the ponto facultativo +/// entries (Carnaval, Corpus Christi) count; the published package defaults it to `true`, and +/// supplying that default is the DX's job. +/// +/// Only the years 1900 to 2099 are supported, the range the holiday rules are stated for. +pub fn is_business_day(value: i64, include_optional: bool) -> bool { + let year = crate::std_date::year_from_days(value); + if !(1900..=2099).contains(&year) { + return false; + } + let weekday = (((value + 3).rem_euclid(7)) + 1); + if ((weekday == 6) || (weekday == 7)) { + return false; + } + for holiday in get_holidays(year).iter() { + if (!include_optional && (holiday.r#type == "optional")) { + continue; + } + if ((holiday.date.cmp(&value) as i64) == 0) { + return false; + } + } + return true; +} diff --git a/core/out/rust/src/is_valid_cnpj.rs b/core/out/rust/src/is_valid_cnpj.rs new file mode 100644 index 000000000..a2739e0a1 --- /dev/null +++ b/core/out/rust/src/is_valid_cnpj.rs @@ -0,0 +1,68 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: is-valid-cnpj +// content: 611130f5f12f + +use crate::*; + +/// Validates a CNPJ (Cadastro Nacional da Pessoa Jurídica), numeric or alphanumeric. +/// +/// Version `"2"` accepts the alphanumeric format as well; a value with no letters is always read +/// as the numeric one, which is also where the reserved repeated numbers are rejected. Mapping a +/// missing or unexpected `options.version` onto `"1"` is the DX's job. +pub fn is_valid_cnpj(cnpj: &str, version: &str) -> bool { + let trimmed = cnpj + .trim_matches(|c: char| { + matches!( + c as u32, + 9 | 10 + | 11 + | 12 + | 13 + | 32 + | 160 + | 5760 + | 8192 + | 8193 + | 8194 + | 8195 + | 8196 + | 8197 + | 8198 + | 8199 + | 8200 + | 8201 + | 8202 + | 8232 + | 8233 + | 8239 + | 8287 + | 12288 + | 65279 + ) + }) + .to_string(); + if (version == "2") { + let cleaned = keep_alphanumeric(cnpj); + if (has_letter(&cleaned) && ((cleaned.len() as i64) == 14)) { + return (crate::support::re_match_1( + &trimmed + .chars() + .map(|c| { + if c.is_ascii_lowercase() { + c.to_ascii_uppercase() + } else { + c + } + }) + .collect::(), + ) && has_valid_cnpj_checksum(&cleaned)); + } + } + let numeric = keep_digits(cnpj); + if ((numeric.len() as i64) != 14) { + return false; + } + return ((crate::support::re_match_2(&trimmed) && !is_repeated_cnpj(&numeric)) + && has_valid_cnpj_checksum(&numeric)); +} diff --git a/core/out/rust/src/is_valid_cpf.rs b/core/out/rust/src/is_valid_cpf.rs new file mode 100644 index 000000000..aa113f4a3 --- /dev/null +++ b/core/out/rust/src/is_valid_cpf.rs @@ -0,0 +1,53 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: is-valid-cpf +// content: dae364e11fb3 + +use crate::*; + +/// Validates a CPF (Cadastro de Pessoas Físicas). +/// +/// The core takes the value as written, accepting the usual mask characters; turning a host value +/// into a string is the DX's job. +pub fn is_valid_cpf(cpf: &str) -> bool { + if !(crate::support::re_match_3(cpf.trim_matches(|c: char| { + matches!( + c as u32, + 9 | 10 + | 11 + | 12 + | 13 + | 32 + | 160 + | 5760 + | 8192 + | 8193 + | 8194 + | 8195 + | 8196 + | 8197 + | 8198 + | 8199 + | 8200 + | 8201 + | 8202 + | 8232 + | 8233 + | 8239 + | 8287 + | 12288 + | 65279 + ) + }))) { + return false; + } + let digits = keep_digits(cpf); + if ((digits.len() as i64) != 11) { + return false; + } + if is_repeated(&digits) { + return false; + } + return ((digit_at_2(&digits) == cpf_check_digit(&digits)) + && (digit_at_3(&digits) == cpf_check_digit_1(&digits))); +} diff --git a/core/out/rust/src/json.rs b/core/out/rust/src/json.rs new file mode 100644 index 000000000..9815ede36 --- /dev/null +++ b/core/out/rust/src/json.rs @@ -0,0 +1,235 @@ +//! A minimal JSON reader and writer for the differential driver's own protocol. This is driver +//! code, not project code: the driver has to speak JSON lines to the conformance harness, and +//! `std` has none, so this is the one piece of machinery it needs that `std` does not supply — +//! not a general-purpose JSON library, just enough to round-trip the protocol's own shapes +//! (numbers, strings, bools, null, arrays and objects of those). + +#[derive(Clone, Debug, PartialEq)] +pub enum Json { + Null, + Bool(bool), + Number(f64), + String(String), + Array(Vec), + Object(Vec<(String, Json)>), +} + +impl Json { + pub fn as_str(&self) -> &str { + match self { + Json::String(s) => s, + _ => "", + } + } + + pub fn as_i64(&self) -> i64 { + match self { + Json::Number(n) => *n as i64, + _ => 0, + } + } + + pub fn as_f64(&self) -> f64 { + match self { + Json::Number(n) => *n, + _ => 0.0, + } + } + + pub fn as_bool(&self) -> bool { + matches!(self, Json::Bool(true)) + } + + pub fn as_array(&self) -> &[Json] { + match self { + Json::Array(items) => items, + _ => &[], + } + } + + pub fn get(&self, key: &str) -> Option<&Json> { + match self { + Json::Object(fields) => fields + .iter() + .find(|(name, _)| name == key) + .map(|(_, value)| value), + _ => None, + } + } +} + +pub fn parse(text: &str) -> Json { + let chars: Vec = text.chars().collect(); + let mut pos = 0usize; + parse_value(&chars, &mut pos) +} + +fn skip_space(chars: &[char], pos: &mut usize) { + while *pos < chars.len() && chars[*pos].is_whitespace() { + *pos += 1; + } +} + +fn parse_value(chars: &[char], pos: &mut usize) -> Json { + skip_space(chars, pos); + match chars.get(*pos) { + Some('{') => parse_object(chars, pos), + Some('[') => parse_array(chars, pos), + Some('"') => Json::String(parse_string(chars, pos)), + Some('t') => { + *pos += 4; + Json::Bool(true) + } + Some('f') => { + *pos += 5; + Json::Bool(false) + } + Some('n') => { + *pos += 4; + Json::Null + } + _ => parse_number(chars, pos), + } +} + +fn parse_object(chars: &[char], pos: &mut usize) -> Json { + *pos += 1; + let mut fields = Vec::new(); + skip_space(chars, pos); + if chars.get(*pos) == Some(&'}') { + *pos += 1; + return Json::Object(fields); + } + loop { + skip_space(chars, pos); + let key = parse_string(chars, pos); + skip_space(chars, pos); + *pos += 1; // ':' + let value = parse_value(chars, pos); + fields.push((key, value)); + skip_space(chars, pos); + match chars.get(*pos) { + Some(',') => { + *pos += 1; + } + _ => { + *pos += 1; // '}' + break; + } + } + } + Json::Object(fields) +} + +fn parse_array(chars: &[char], pos: &mut usize) -> Json { + *pos += 1; + let mut items = Vec::new(); + skip_space(chars, pos); + if chars.get(*pos) == Some(&']') { + *pos += 1; + return Json::Array(items); + } + loop { + let value = parse_value(chars, pos); + items.push(value); + skip_space(chars, pos); + match chars.get(*pos) { + Some(',') => { + *pos += 1; + } + _ => { + *pos += 1; // ']' + break; + } + } + } + Json::Array(items) +} + +fn parse_string(chars: &[char], pos: &mut usize) -> String { + *pos += 1; // opening quote + let mut out = String::new(); + while let Some(&c) = chars.get(*pos) { + *pos += 1; + if c == '"' { + break; + } + if c == '\\' { + let escaped = chars.get(*pos).copied().unwrap_or('\\'); + *pos += 1; + match escaped { + 'n' => out.push('\n'), + 'r' => out.push('\r'), + 't' => out.push('\t'), + 'u' => { + let hex: String = chars[*pos..*pos + 4].iter().collect(); + *pos += 4; + if let Ok(code) = u32::from_str_radix(&hex, 16) { + if let Some(scalar) = char::from_u32(code) { + out.push(scalar); + } + } + } + other => out.push(other), + } + } else { + out.push(c); + } + } + out +} + +fn parse_number(chars: &[char], pos: &mut usize) -> Json { + let start = *pos; + while chars.get(*pos).is_some_and(|c| { + c.is_ascii_digit() || *c == '-' || *c == '+' || *c == '.' || *c == 'e' || *c == 'E' + }) { + *pos += 1; + } + let text: String = chars[start..*pos].iter().collect(); + Json::Number(text.parse::().unwrap_or(0.0)) +} + +pub fn write(value: &Json) -> String { + match value { + Json::Null => "null".to_string(), + Json::Bool(b) => b.to_string(), + Json::Number(n) => { + if n.fract() == 0.0 && n.abs() < 1e15 { + format!("{}", *n as i64) + } else { + format!("{}", n) + } + } + Json::String(s) => write_string(s), + Json::Array(items) => format!( + "[{}]", + items.iter().map(write).collect::>().join(",") + ), + Json::Object(fields) => format!( + "{{{}}}", + fields + .iter() + .map(|(key, value)| format!("{}:{}", write_string(key), write(value))) + .collect::>() + .join(",") + ), + } +} + +fn write_string(value: &str) -> String { + let mut out = String::from("\""); + for c in value.chars() { + match c { + '"' => out.push_str("\\\""), + '\\' => out.push_str("\\\\"), + '\n' => out.push_str("\\n"), + '\r' => out.push_str("\\r"), + '\t' => out.push_str("\\t"), + c if (c as u32) < 0x20 => out.push_str(&format!("\\u{:04x}", c as u32)), + c => out.push(c), + } + } + out.push('"'); + out +} diff --git a/core/out/rust/src/lib.rs b/core/out/rust/src/lib.rs new file mode 100644 index 000000000..1eb340190 --- /dev/null +++ b/core/out/rust/src/lib.rs @@ -0,0 +1,62 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib +// +// `needless_return`: every function body ends with an explicit `return`, mirroring the +// Core's own structure (see the module comment on ownership); a tail-position rewrite would +// have to reconstruct reachability through `if`/`match`/loops that the Core already settled. +// `unused_parens`: the printer parenthesizes every binary and ternary uniformly rather than +// tracking each operator's precedence, which is what makes the printer itself precedence-free. +// `vec_init_then_push`: `push` on a local list, one call per element, is the Core's own +// accumulation idiom (`docs/semantics.md` names it explicitly); collapsing a run of them into +// one `vec![...]` literal would need the printer to prove nothing between them can fail or +// branch, which a source author's control flow can defeat in ways this backend does not chase. +#![allow( + clippy::needless_return, + clippy::vec_init_then_push, + unused_parens, + unused_imports +)] + +pub mod errors; +pub mod format_cnpj; +pub mod format_currency; +pub mod generate_cnpj; +pub mod generate_cpf; +pub mod get_address_info_by_cep; +pub mod get_holidays; +pub mod is_business_day; +pub mod is_valid_cnpj; +pub mod is_valid_cpf; +pub mod json; +pub mod lib_civil; +pub mod lib_cnpj; +pub mod lib_cpf; +pub mod lib_digits; +pub mod lib_easter; +pub mod lib_format; +pub mod lib_json; +pub mod lib_random; +pub mod std_date; +pub mod support; + +pub use errors::*; +pub use format_cnpj::*; +pub use format_currency::*; +pub use generate_cnpj::*; +pub use generate_cpf::*; +pub use get_address_info_by_cep::*; +pub use get_holidays::*; +pub use is_business_day::*; +pub use is_valid_cnpj::*; +pub use is_valid_cpf::*; +pub use lib_civil::*; +pub use lib_cnpj::*; +pub use lib_cpf::*; +pub use lib_digits::*; +pub use lib_easter::*; +pub use lib_format::*; +pub use lib_json::*; +pub use lib_random::*; +pub use std_date::*; +pub use support::*; diff --git a/core/out/rust/src/lib_civil.rs b/core/out/rust/src/lib_civil.rs new file mode 100644 index 000000000..e022a9fe0 --- /dev/null +++ b/core/out/rust/src/lib_civil.rs @@ -0,0 +1,61 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/civil +// content: c2aca1eb50dc + +use crate::*; + +/// A fixed day of a year, with the unreachable fallback named once. +pub(crate) fn civil_date(year: i64) -> i64 { + return crate::std_date::ymd_to_days(year, 1, 1).unwrap_or(0.clamp(-719162, 2932896)); +} + +/// A fixed day of a year, with the unreachable fallback named once. +pub(crate) fn civil_date_1(year: i64) -> i64 { + return crate::std_date::ymd_to_days(year, 4, 21).unwrap_or(0.clamp(-719162, 2932896)); +} + +/// A fixed day of a year, with the unreachable fallback named once. +pub(crate) fn civil_date_2(year: i64) -> i64 { + return crate::std_date::ymd_to_days(year, 5, 1).unwrap_or(0.clamp(-719162, 2932896)); +} + +/// A fixed day of a year, with the unreachable fallback named once. +pub(crate) fn civil_date_3(year: i64) -> i64 { + return crate::std_date::ymd_to_days(year, 9, 7).unwrap_or(0.clamp(-719162, 2932896)); +} + +/// A fixed day of a year, with the unreachable fallback named once. +pub(crate) fn civil_date_4(year: i64) -> i64 { + return crate::std_date::ymd_to_days(year, 10, 12).unwrap_or(0.clamp(-719162, 2932896)); +} + +/// A fixed day of a year, with the unreachable fallback named once. +pub(crate) fn civil_date_5(year: i64) -> i64 { + return crate::std_date::ymd_to_days(year, 11, 2).unwrap_or(0.clamp(-719162, 2932896)); +} + +/// A fixed day of a year, with the unreachable fallback named once. +pub(crate) fn civil_date_6(year: i64) -> i64 { + return crate::std_date::ymd_to_days(year, 11, 15).unwrap_or(0.clamp(-719162, 2932896)); +} + +/// A fixed day of a year, with the unreachable fallback named once. +pub(crate) fn civil_date_7(year: i64) -> i64 { + return crate::std_date::ymd_to_days(year, 12, 25).unwrap_or(0.clamp(-719162, 2932896)); +} + +/// A fixed day of a year, with the unreachable fallback named once. +pub(crate) fn civil_date_8(year: i64) -> i64 { + return crate::std_date::ymd_to_days(year, 11, 20).unwrap_or(0.clamp(-719162, 2932896)); +} + +/// A fixed day of a year, with the unreachable fallback named once. +pub(crate) fn civil_date_9(year: i64, day: i64) -> i64 { + return crate::std_date::ymd_to_days(year, 3, day).unwrap_or(0.clamp(-719162, 2932896)); +} + +/// A fixed day of a year, with the unreachable fallback named once. +pub(crate) fn civil_date_10(year: i64, day: i64) -> i64 { + return crate::std_date::ymd_to_days(year, 4, day).unwrap_or(0.clamp(-719162, 2932896)); +} diff --git a/core/out/rust/src/lib_cnpj.rs b/core/out/rust/src/lib_cnpj.rs new file mode 100644 index 000000000..9e953882b --- /dev/null +++ b/core/out/rust/src/lib_cnpj.rs @@ -0,0 +1,100 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/cnpj +// content: 78a3fa4783b9 + +use crate::*; + +pub const LIB_CNPJ_TABLE1: &[i64] = &[5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]; + +pub const LIB_CNPJ_TABLE2: &[i64] = &[6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]; + +/// A random numeric CNPJ base: an 8-digit root and a 4-digit branch, each digit drawn +/// independently — matches the published `generateCnpj()` called with no branch, where an unset +/// branch also draws those 4 digits at random. Twelve separate draws, not a loop, is what lets the +/// result stay exactly 12 digits long. +pub(crate) fn random_cnpj_base(env: &dyn Capabilities) -> String { + return { + let __piece0 = random_digit(env); + let __piece1 = random_digit(env); + let __piece2 = random_digit(env); + let __piece3 = random_digit(env); + let __piece4 = random_digit(env); + let __piece5 = random_digit(env); + let __piece6 = random_digit(env); + let __piece7 = random_digit(env); + let __piece8 = random_digit(env); + let __piece9 = random_digit(env); + let __piece10 = random_digit(env); + let __piece11 = random_digit(env); + let mut __buf = String::with_capacity( + __piece0.len() + + __piece1.len() + + __piece2.len() + + __piece3.len() + + __piece4.len() + + __piece5.len() + + __piece6.len() + + __piece7.len() + + __piece8.len() + + __piece9.len() + + __piece10.len() + + __piece11.len(), + ); + __buf.push_str(&__piece0); + __buf.push_str(&__piece1); + __buf.push_str(&__piece2); + __buf.push_str(&__piece3); + __buf.push_str(&__piece4); + __buf.push_str(&__piece5); + __buf.push_str(&__piece6); + __buf.push_str(&__piece7); + __buf.push_str(&__piece8); + __buf.push_str(&__piece9); + __buf.push_str(&__piece10); + __buf.push_str(&__piece11); + __buf + }; +} + +/// The check digit of a CNPJ base, under the rule both versions share. +pub(crate) fn cnpj_check_digit(cnpj: &str, weights: &[i64]) -> i64 { + let mut sum = 0; + for index in 0..(weights.len() as i64) { + sum += (((cnpj.as_bytes()[index as usize] as i64) - 48) + * crate::support::get_at(weights, index)); + } + let remainder = (sum % 11); + return (if (remainder < 2) { 0 } else { (11 - remainder) }); +} + +/// Whether the value holds at least one upper cased ASCII letter. +/// +/// The scan reads positions rather than materializing the scalars, which the checked accessor +/// makes safe without a proof about the length. +pub(crate) fn has_letter(value: &str) -> bool { + for index in 0..(value.len() as i64) { + let point = crate::support::code_at(value, index).unwrap_or(0); + if (65..=90).contains(&point) { + return true; + } + } + return false; +} + +/// Whether both check digits of a 14 character CNPJ match its base. +pub(crate) fn has_valid_cnpj_checksum(cnpj: &str) -> bool { + return ((((cnpj.as_bytes()[12] as i64) - 48) == cnpj_check_digit(cnpj, LIB_CNPJ_TABLE1)) + && (((cnpj.as_bytes()[13] as i64) - 48) == cnpj_check_digit(cnpj, LIB_CNPJ_TABLE2))); +} + +/// Whether every character of a 14 character value is the same one. +pub(crate) fn is_repeated_cnpj(value: &str) -> bool { + let first = (value.as_bytes()[0] as i64); + for index in 1..14 { + if ((value.as_bytes()[index as usize] as i64) != first) { + return false; + } + } + return true; +} diff --git a/core/out/rust/src/lib_cpf.rs b/core/out/rust/src/lib_cpf.rs new file mode 100644 index 000000000..ce1c3d2b8 --- /dev/null +++ b/core/out/rust/src/lib_cpf.rs @@ -0,0 +1,75 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/cpf +// content: 5b319831cfe4 + +use crate::*; + +/// A random CPF base: 8 digits plus a região fiscal digit, each drawn independently — matches the +/// published `generateCpf()` called with no state, where an unset state also draws that 9th digit +/// at random. Nine separate draws, not a loop, is what lets the result stay exactly 9 digits long. +pub(crate) fn random_cpf_base(env: &dyn Capabilities) -> String { + return { + let __piece0 = random_digit(env); + let __piece1 = random_digit(env); + let __piece2 = random_digit(env); + let __piece3 = random_digit(env); + let __piece4 = random_digit(env); + let __piece5 = random_digit(env); + let __piece6 = random_digit(env); + let __piece7 = random_digit(env); + let __piece8 = random_digit(env); + let mut __buf = String::with_capacity( + __piece0.len() + + __piece1.len() + + __piece2.len() + + __piece3.len() + + __piece4.len() + + __piece5.len() + + __piece6.len() + + __piece7.len() + + __piece8.len(), + ); + __buf.push_str(&__piece0); + __buf.push_str(&__piece1); + __buf.push_str(&__piece2); + __buf.push_str(&__piece3); + __buf.push_str(&__piece4); + __buf.push_str(&__piece5); + __buf.push_str(&__piece6); + __buf.push_str(&__piece7); + __buf.push_str(&__piece8); + __buf + }; +} + +/// The check digit of a CPF base, under the Receita Federal rule (weights 10..2 and 11..2). +pub(crate) fn cpf_check_digit(cpf: &str) -> i64 { + let mut sum = 0; + for index in 0..9 { + sum += (digit_at(cpf, index) * (10 - index)); + } + let remainder = (sum % 11); + return (if (remainder < 2) { 0 } else { (11 - remainder) }); +} + +/// The check digit of a CPF base, under the Receita Federal rule (weights 10..2 and 11..2). +pub(crate) fn cpf_check_digit_1(cpf: &str) -> i64 { + let mut sum = 0; + for index in 0..10 { + sum += (digit_at(cpf, index) * (11 - index)); + } + let remainder = (sum % 11); + return (if (remainder < 2) { 0 } else { (11 - remainder) }); +} + +/// Whether every scalar of the value is the same one, e.g. "00000000000". +pub(crate) fn is_repeated(value: &str) -> bool { + let first = (value.as_bytes()[0] as i64); + for index in 1..11 { + if ((value.as_bytes()[index as usize] as i64) != first) { + return false; + } + } + return true; +} diff --git a/core/out/rust/src/lib_digits.rs b/core/out/rust/src/lib_digits.rs new file mode 100644 index 000000000..0aba1a9bb --- /dev/null +++ b/core/out/rust/src/lib_digits.rs @@ -0,0 +1,61 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/digits +// content: 780bb464b89e + +use crate::*; + +/// Keeps only the ASCII digits and letters of a value, upper casing the letters. +pub(crate) fn keep_alphanumeric(value: &str) -> String { + return { + let mut __out = String::with_capacity(value.len()); + for __b in value.bytes() { + if (48..=57).contains(&__b) || (65..=90).contains(&__b) || (97..=122).contains(&__b) { + __out.push(__b as char); + } + } + __out + } + .to_ascii_uppercase(); +} + +/// Keeps only the ASCII digits of a value, dropping every mask character. +pub(crate) fn keep_digits(value: &str) -> String { + return { + let mut __out = String::with_capacity(value.len()); + for __b in value.bytes() { + if (48..=57).contains(&__b) { + __out.push(__b as char); + } + } + __out + }; +} + +/// Whether every scalar of the value is the same one, for whatever length the caller proved — +/// `isRepeated` and `isRepeatedCnpj` do the same check for one specific length; this one serves a +/// generator that has to run it on a base shorter than the document it is building. +pub(crate) fn is_repeated_run(value: &str) -> bool { + let first = (value.as_bytes()[0] as i64); + for index in 1..(value.len() as i64) { + if ((value.as_bytes()[index as usize] as i64) != first) { + return false; + } + } + return true; +} + +/// The numeric value of one ASCII digit. +pub(crate) fn digit_at(value: &str, index: i64) -> i64 { + return ((value.as_bytes()[index as usize] as i64) - 48); +} + +/// The numeric value of one ASCII digit. +pub(crate) fn digit_at_2(value: &str) -> i64 { + return ((value.as_bytes()[9] as i64) - 48); +} + +/// The numeric value of one ASCII digit. +pub(crate) fn digit_at_3(value: &str) -> i64 { + return ((value.as_bytes()[10] as i64) - 48); +} diff --git a/core/out/rust/src/lib_easter.rs b/core/out/rust/src/lib_easter.rs new file mode 100644 index 000000000..5c7824059 --- /dev/null +++ b/core/out/rust/src/lib_easter.rs @@ -0,0 +1,32 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/easter +// content: c99e5cf5c72b + +use crate::*; + +/// The day of March (1 to 31) or April (32 to 56) Easter falls on, as a day-of-March offset. +pub(crate) fn easter_day_of_march(year: i64) -> i64 { + let a = (year % 19); + let b = (year / 100); + let c = (year % 100); + let d = (b / 4); + let e = (b % 4); + let h = ((((((19 * a) + b) - d) - 6) + 15) % 30); + let i = (c / 4); + let k = (c % 4); + let l = (((((32 + (2 * e)) + (2 * i)) - h) - k) % 7); + let m = (((a + (11 * h)) + (22 * l)) / 451); + let day = (((h + l) - (7 * m)) + 114); + return (((day % 31) + 1) + (((day / 31) - 3) * 31)).clamp(22, 56); +} + +/// Easter Sunday of a year, as a civil date. +pub(crate) fn easter_sunday(year: i64) -> i64 { + let day_of_march = easter_day_of_march(year); + return (if (day_of_march <= 31) { + civil_date_9(year, day_of_march) + } else { + civil_date_10(year, (day_of_march - 31)) + }); +} diff --git a/core/out/rust/src/lib_format.rs b/core/out/rust/src/lib_format.rs new file mode 100644 index 000000000..59b30f459 --- /dev/null +++ b/core/out/rust/src/lib_format.rs @@ -0,0 +1,71 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/format +// content: 0038a81c2e4d + +use crate::*; + +/// How many scalars of the value a pattern consumes. +pub(crate) fn pattern_slots(pattern: &str) -> i64 { + let mut slots = 0; + for index in 0..(pattern.len() as i64) { + let symbol = crate::support::char_at(pattern, index).unwrap_or("".to_string()); + if ((symbol == "0") || (symbol == "*")) { + slots += 1; + } + } + return slots; +} + +/// Formats a value against a pattern, optionally left padding it with zeros first. +pub(crate) fn format_with_pattern(value: &str, pattern: &str, pad: bool) -> String { + let padded = (if pad { + crate::support::pad_start_ascii(value, pattern_slots(pattern), "0") + } else { + value.to_owned() + }); + let mut out = "".to_string(); + let mut taken = 0; + for index in 0..(pattern.len() as i64) { + let symbol = crate::support::char_at(pattern, index).unwrap_or("".to_string()); + if ((symbol == "0") || (symbol == "*")) { + if (taken >= (padded.len() as i64)) { + return out.to_owned(); + } + out = crate::support::concat2( + &out, + &(if (symbol == "*") { + "*".to_string() + } else { + crate::support::char_at(&padded, taken).unwrap_or("".to_string()) + }), + ); + taken += 1; + } else { + if (taken < (padded.len() as i64)) { + out = crate::support::concat2(&out, &symbol); + } + } + } + return out.to_owned(); +} + +/// Groups the whole part with `.` every three digits, the pt-BR convention. +pub(crate) fn group_thousands(whole: &str) -> String { + let mut out = vec![]; + let scalars = whole.bytes().map(|b| b as i64).collect::>(); + for index in 0..(scalars.len() as i64) { + if ((index > 0) && ((((scalars.len() as i64) - index) % 3) == 0)) { + out.push(46); + } + out.push(crate::support::at(&scalars, index).unwrap_or(48)); + } + return { + let __pts = &out; + let mut __out = String::with_capacity(__pts.len()); + for &__p in __pts { + __out.push(__p as u8 as char); + } + __out + }; +} diff --git a/core/out/rust/src/lib_json.rs b/core/out/rust/src/lib_json.rs new file mode 100644 index 000000000..49a746822 --- /dev/null +++ b/core/out/rust/src/lib_json.rs @@ -0,0 +1,127 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/json +// content: 02cc75dfd626 + +use crate::*; + +/// Whether `needle` occurs in `points` at `start`. +fn matches_at(points: &[i64], needle: &[i64], start: i64) -> bool { + for offset in 0..(needle.len() as i64) { + if (crate::support::at(points, (start + offset)).unwrap_or(-1) + != crate::support::at(needle, offset).unwrap_or(-2)) + { + return false; + } + } + return true; +} + +/// Whether a code point is JSON whitespace. +fn is_space(point: i64) -> bool { + return ((((point == 32) || (point == 9)) || (point == 10)) || (point == 13)); +} + +/// The hexadecimal value of four scalars, for a `\uXXXX` escape. +fn hex_value(points: &[i64], start: i64) -> i64 { + let mut value = 0; + for offset in 0..4 { + let point = crate::support::at(points, (start + offset)).unwrap_or(48); + let mut digit = 0; + if (48..=57).contains(&point) { + digit = (point - 48); + } else { + if (97..=102).contains(&point) { + digit = (point - 87); + } else { + if (65..=70).contains(&point) { + digit = (point - 55); + } + } + } + value = ((value * 16) + digit); + } + return value.min(65535); +} + +/// The string value of a top-level JSON field, or absent when the field is missing or is not a +/// string. Escapes are decoded; a surrogate pair is left as its two escaped halves, which no CEP +/// provider emits. +pub(crate) fn json_string_field(body: &str, key: &str) -> Option { + let points = crate::support::code_points(body); + let needle = { + let mut __buf = String::with_capacity("\"".len() + key.len() + "\"".len()); + __buf.push('"'); + __buf.push_str(key); + __buf.push('"'); + __buf + } + .bytes() + .map(|b| b as i64) + .collect::>(); + for index in 0..(points.len() as i64) { + if !matches_at(&points, &needle, index) { + continue; + } + let mut cursor = (index + (needle.len() as i64)); + for _skip in 0..8 { + if is_space(crate::support::at(&points, cursor).unwrap_or(0)) { + cursor += 1; + } + } + if (crate::support::at(&points, cursor).unwrap_or(0) != 58) { + continue; + } + cursor += 1; + for _skip in 0..8 { + if is_space(crate::support::at(&points, cursor).unwrap_or(0)) { + cursor += 1; + } + } + if (crate::support::at(&points, cursor).unwrap_or(0) != 34) { + continue; + } + cursor += 1; + let mut out = vec![]; + for _step in 0..(points.len() as i64) { + let point = crate::support::at(&points, cursor).unwrap_or(-1); + if ((point == -1) || (point == 34)) { + return Some(crate::support::from_code_points(&out)); + } + if (point == 92) { + let escaped = crate::support::at(&points, (cursor + 1)).unwrap_or(-1); + if (escaped == 110) { + out.push(10); + cursor += 2; + } else { + if (escaped == 116) { + out.push(9); + cursor += 2; + } else { + if (escaped == 114) { + out.push(13); + cursor += 2; + } else { + if (escaped == 117) { + out.push(hex_value(&points, (cursor + 2))); + cursor += 6; + } else { + if (escaped >= 0) { + out.push(escaped.to_owned()); + cursor += 2; + } else { + cursor += 1; + } + } + } + } + } + } else { + out.push(point.to_owned()); + cursor += 1; + } + } + return Some(crate::support::from_code_points(&out)); + } + return None; +} diff --git a/core/out/rust/src/lib_random.rs b/core/out/rust/src/lib_random.rs new file mode 100644 index 000000000..c741e18cb --- /dev/null +++ b/core/out/rust/src/lib_random.rs @@ -0,0 +1,26 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/random +// content: 2101f5409061 + +use crate::*; + +/// A uniform integer in `[0, bound)`, by rejection sampling rather than `% bound`: the modulo of a +/// fixed-width draw is biased whenever `bound` does not divide 2^32 evenly, and that bias would +/// have to match, digit for digit, across three unrelated standard libraries to stay invisible. +/// Rejecting the biased tail of the draw removes it instead. +pub(crate) fn random_below(env: &dyn Capabilities) -> i64 { + let limit = 4294967290; + for _attempt in 0..32 { + let draw = env.next_u32(); + if (draw < limit) { + return (draw % 10); + } + } + return (env.next_u32() % 10); +} + +/// One random ASCII digit. +pub(crate) fn random_digit(env: &dyn Capabilities) -> String { + return crate::support::digit_char(random_below(env)); +} diff --git a/core/out/rust/src/std_date.rs b/core/out/rust/src/std_date.rs new file mode 100644 index 000000000..63f5fa9e2 --- /dev/null +++ b/core/out/rust/src/std_date.rs @@ -0,0 +1,121 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: std/date +// content: 8041c981a090 + +use crate::*; + +/// Floor division, which the calendar algorithms need for negative years. +fn floor_div_1(value: i64) -> i64 { + let quotient = (value / 400); + if ((value < 0) && ((quotient * 400) != value)) { + return (quotient - 1); + } + return quotient; +} + +/// Days since 1970-01-01 for a year, month and day already known to be a real date. +pub(crate) fn days_from_civil(year: i64, month: i64, day: i64) -> i64 { + let shifted = (if (month <= 2) { + (year - 1) + } else { + year.to_owned() + }); + let era = floor_div_1(shifted); + let year_of_era = (shifted - (era * 400)); + let month_term = (if (month > 2) { + (month - 3) + } else { + (month + 9) + }); + let day_of_year = (((((153 * month_term) + 2) / 5) + day) - 1); + let day_of_era = + ((((year_of_era * 365) + (year_of_era / 4)) - (year_of_era / 100)) + day_of_year); + return (((era * 146097) + day_of_era) - 719468).clamp(-719162, 2932896); +} + +/// Floor division, which the calendar algorithms need for negative years. +pub(crate) fn floor_div(value: i64) -> i64 { + let quotient = (value / 146097); + if ((value < 0) && ((quotient * 146097) != value)) { + return (quotient - 1); + } + return quotient; +} + +/// The year of a date given as days since 1970-01-01. +pub(crate) fn year_from_days(days: i64) -> i64 { + let shifted = (days + 719468); + let era = floor_div(shifted); + let day_of_era = (shifted - (era * 146097)); + let year_of_era = ((((day_of_era - (day_of_era / 1460)) + (day_of_era / 36524)) + - (day_of_era / 146096)) + / 365); + let year = (year_of_era + (era * 400)); + let day_of_year = + (day_of_era - (((365 * year_of_era) + (year_of_era / 4)) - (year_of_era / 100))); + let month_prime = (((5 * day_of_year) + 2) / 153); + let month = (if (month_prime < 10) { + (month_prime + 3) + } else { + (month_prime - 9) + }); + return (if (month <= 2) { + (year + 1) + } else { + year.to_owned() + }) + .clamp(1, 9999); +} + +/// The month of a date given as days since 1970-01-01. +pub(crate) fn month_from_days(days: i64) -> i64 { + let shifted = (days + 719468); + let era = floor_div(shifted); + let day_of_era = (shifted - (era * 146097)); + let year_of_era = ((((day_of_era - (day_of_era / 1460)) + (day_of_era / 36524)) + - (day_of_era / 146096)) + / 365); + let day_of_year = + (day_of_era - (((365 * year_of_era) + (year_of_era / 4)) - (year_of_era / 100))); + let month_prime = (((5 * day_of_year) + 2) / 153); + return (if (month_prime < 10) { + (month_prime + 3) + } else { + (month_prime - 9) + }) + .clamp(1, 12); +} + +/// The day of month of a date given as days since 1970-01-01. +pub(crate) fn day_from_days(days: i64) -> i64 { + let shifted = (days + 719468); + let era = floor_div(shifted); + let day_of_era = (shifted - (era * 146097)); + let year_of_era = ((((day_of_era - (day_of_era / 1460)) + (day_of_era / 36524)) + - (day_of_era / 146096)) + / 365); + let day_of_year = + (day_of_era - (((365 * year_of_era) + (year_of_era / 4)) - (year_of_era / 100))); + let month_prime = (((5 * day_of_year) + 2) / 153); + return ((day_of_year - (((153 * month_prime) + 2) / 5)) + 1).clamp(1, 31); +} + +/// Days since 1970-01-01, or absent when the components do not name a real date. +/// +/// The bounds are checked here rather than in a helper because the checker reads a guard, not a +/// called predicate: after this `if`, the three components carry the ranges `daysFromCivil` +/// requires, and the round trip rejects a day the month does not have. +pub(crate) fn ymd_to_days(year: i64, month: i64, day: i64) -> Option { + if ((((!(1..=9999).contains(&year) || (month < 1)) || (month > 12)) || (day < 1)) || (day > 31)) + { + return None; + } + let days = days_from_civil(year, month, day); + if (((year_from_days(days) != year) || (month_from_days(days) != month)) + || (day_from_days(days) != day)) + { + return None; + } + return Some(days.to_owned()); +} diff --git a/core/out/rust/src/support.rs b/core/out/rust/src/support.rs new file mode 100644 index 000000000..972592b1a --- /dev/null +++ b/core/out/rust/src/support.rs @@ -0,0 +1,516 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: support +#![allow(dead_code)] + +pub fn concat2(a: &str, b: &str) -> String { + let mut out = String::with_capacity(a.len() + b.len()); + out.push_str(a); + out.push_str(b); + out +} + +pub fn code_points(value: &str) -> Vec { + value.chars().map(|c| c as i64).collect() +} + +pub fn from_code_points(points: &[i64]) -> String { + points + .iter() + .map(|&p| char::from_u32(p as u32).unwrap_or('\u{fffd}')) + .collect() +} + +pub fn as_ascii(value: &str) -> Option { + if value.chars().all(|c| (c as u32) < 0x80) { + Some(value.to_string()) + } else { + None + } +} + +pub fn as_digits(value: &str) -> Option { + if !value.is_empty() && value.chars().all(|c| c.is_ascii_digit()) { + Some(value.to_string()) + } else { + None + } +} + +pub fn parse_digits(value: &str) -> Option { + if value.is_empty() || value.len() > 18 || !value.chars().all(|c| c.is_ascii_digit()) { + return None; + } + value.parse::().ok() +} + +/// Consumes exactly `count` chars matching `in_class` off the front of `rest`, or answers `None` +/// without consuming anything. One forward pass, no allocation: this and `re_take_class` below are +/// the whole of a generated chain-pattern scanner (`re_match_N`, in the "Regex" section of +/// engine/src/targets/rust/index.ts) — a fixed-count class run in the pattern becomes one call here. +/// Every class this project matches against is ASCII except the mask-separator whitespace class, +/// and even that one is ASCII on almost every byte a real caller passes (plain digits, or digits +/// plus '.', '-', '/' and ' ' -- see `core/source`'s own patterns) -- so the leading byte is tested +/// directly first; only a byte that starts a multi-byte sequence pays for decoding a full `char`. +#[inline] +fn re_take_fixed(rest: &str, count: usize, in_class: impl Fn(u32) -> bool) -> Option<&str> { + let bytes = rest.as_bytes(); + let mut pos = 0usize; + let mut taken = 0usize; + while taken < count { + let &b = bytes.get(pos)?; + if b < 0x80 { + if !in_class(b as u32) { + return None; + } + pos += 1; + } else { + let ch = rest[pos..].chars().next().unwrap(); + if !in_class(ch as u32) { + return None; + } + pos += ch.len_utf8(); + } + taken += 1; + } + Some(&rest[pos..]) +} + +/// Consumes as many chars matching `in_class` as `rest` offers, up to `max` (`usize::MAX` for +/// unbounded), then answers `None` unless at least `min` were taken. The maximal-munch property +/// `chainElementsOf` checks at generation time (see the "Regex" section) is what makes always +/// taking the longest available run — never backing off to try a shorter one — correct here. Same +/// ASCII-first byte test as `re_take_fixed` above, for the same reason. +#[inline] +fn re_take_class( + rest: &str, + min: usize, + max: usize, + in_class: impl Fn(u32) -> bool, +) -> Option<&str> { + let bytes = rest.as_bytes(); + let mut pos = 0usize; + let mut taken = 0usize; + while taken < max { + let Some(&b) = bytes.get(pos) else { + break; + }; + if b < 0x80 { + if !in_class(b as u32) { + break; + } + pos += 1; + } else { + let ch = rest[pos..].chars().next().unwrap(); + if !in_class(ch as u32) { + break; + } + pos += ch.len_utf8(); + } + taken += 1; + } + if taken < min { + return None; + } + Some(&rest[pos..]) +} + +/// The single-digit fast path `str.fromInt` prefers when the value is proven to be one decimal +/// digit (see the candidate's own comment in engine/src/targets/rust/index.ts): one ASCII byte +/// pushed into a one-byte-capacity `String` is the whole job, no general integer formatter needed. +pub fn digit_char(n: i64) -> String { + let mut out = String::with_capacity(1); + out.push((n as u8 + b'0') as char); + out +} + +/// The ASCII-only fast path `str.padStart` prefers when both `value` and `pad` are proven ASCII +/// (see the candidate's own comment in engine/src/targets/rust/index.ts): a scalar is a byte, so +/// the length check is `value.len()` and each missing slot is `pad` pushed wholesale, with no +/// `Vec` built anywhere -- `pad_start` below builds two just to learn what this already +/// knows. +pub fn pad_start_ascii(value: &str, length: i64, pad: &str) -> String { + let length = length as usize; + if value.len() >= length { + return value.to_string(); + } + let missing = length - value.len(); + let mut out = String::with_capacity(pad.len() * missing + value.len()); + for _ in 0..missing { + out.push_str(pad); + } + out.push_str(value); + out +} + +pub fn pad_start(value: &str, length: i64, pad: &str) -> String { + let scalars: Vec = value.chars().collect(); + let length = length as usize; + if scalars.len() >= length { + return value.to_string(); + } + let pad_scalars: Vec = pad.chars().collect(); + let mut prefix = String::new(); + for _ in 0..(length - scalars.len()) { + prefix.extend(pad_scalars.iter()); + } + prefix + value +} + +pub fn code_at(value: &str, index: i64) -> Option { + let bytes = value.as_bytes(); + if index < 0 || index as usize >= bytes.len() { + return None; + } + Some(bytes[index as usize] as i64) +} + +pub fn char_at(value: &str, index: i64) -> Option { + let bytes = value.as_bytes(); + if index < 0 || index as usize >= bytes.len() { + return None; + } + Some((bytes[index as usize] as char).to_string()) +} + +pub fn at(values: &[T], index: i64) -> Option { + if index < 0 { + return None; + } + values.get(index as usize).cloned() +} + +pub fn get_at(values: &[T], index: i64) -> T { + values[index as usize].clone() +} + +pub fn mapped(values: &[T], f: impl Fn(&T) -> R) -> Vec { + values.iter().map(f).collect() +} + +pub fn filtered(values: &[T], keep: impl Fn(&T) -> bool) -> Vec { + values.iter().filter(|v| keep(v)).cloned().collect() +} + +pub fn sorted_stable(values: &[T]) -> Vec { + let mut out = values.to_vec(); + out.sort(); + out +} + +pub fn sorted_stable_by(values: &[T], key: impl Fn(&T) -> K) -> Vec { + let mut out = values.to_vec(); + out.sort_by_key(key); + out +} + +pub fn date_from_epoch_days(days: i64) -> Option { + if !(-719162..=2932896).contains(&days) { + return None; + } + Some(days) +} + +pub fn re_match_0(value: &str) -> bool { + let rest = value; + let Some(rest) = re_take_fixed(rest, 8, |c: u32| (48..=57).contains(&c)) else { + return false; + }; + rest.is_empty() +} +pub fn re_match_1(value: &str) -> bool { + let rest = value; + let Some(rest) = re_take_fixed(rest, 2, |c: u32| { + (48..=57).contains(&c) || (65..=90).contains(&c) + }) else { + return false; + }; + let Some(rest) = re_take_class(rest, 0, usize::MAX, |c: u32| { + (9..=13).contains(&c) + || c == 32 + || (45..=47).contains(&c) + || c == 160 + || c == 5760 + || (8192..=8202).contains(&c) + || (8232..=8233).contains(&c) + || c == 8239 + || c == 8287 + || c == 12288 + || c == 65279 + }) else { + return false; + }; + let Some(rest) = re_take_fixed(rest, 3, |c: u32| { + (48..=57).contains(&c) || (65..=90).contains(&c) + }) else { + return false; + }; + let Some(rest) = re_take_class(rest, 0, usize::MAX, |c: u32| { + (9..=13).contains(&c) + || c == 32 + || (45..=47).contains(&c) + || c == 160 + || c == 5760 + || (8192..=8202).contains(&c) + || (8232..=8233).contains(&c) + || c == 8239 + || c == 8287 + || c == 12288 + || c == 65279 + }) else { + return false; + }; + let Some(rest) = re_take_fixed(rest, 3, |c: u32| { + (48..=57).contains(&c) || (65..=90).contains(&c) + }) else { + return false; + }; + let Some(rest) = re_take_class(rest, 0, usize::MAX, |c: u32| { + (9..=13).contains(&c) + || c == 32 + || (45..=47).contains(&c) + || c == 160 + || c == 5760 + || (8192..=8202).contains(&c) + || (8232..=8233).contains(&c) + || c == 8239 + || c == 8287 + || c == 12288 + || c == 65279 + }) else { + return false; + }; + let Some(rest) = re_take_fixed(rest, 4, |c: u32| { + (48..=57).contains(&c) || (65..=90).contains(&c) + }) else { + return false; + }; + let Some(rest) = re_take_class(rest, 0, usize::MAX, |c: u32| { + (9..=13).contains(&c) + || c == 32 + || (45..=47).contains(&c) + || c == 160 + || c == 5760 + || (8192..=8202).contains(&c) + || (8232..=8233).contains(&c) + || c == 8239 + || c == 8287 + || c == 12288 + || c == 65279 + }) else { + return false; + }; + let Some(rest) = re_take_fixed(rest, 2, |c: u32| (48..=57).contains(&c)) else { + return false; + }; + rest.is_empty() +} +pub fn re_match_2(value: &str) -> bool { + let rest = value; + let Some(rest) = re_take_fixed(rest, 2, |c: u32| (48..=57).contains(&c)) else { + return false; + }; + let Some(rest) = re_take_class(rest, 0, usize::MAX, |c: u32| { + (9..=13).contains(&c) + || c == 32 + || (45..=47).contains(&c) + || c == 160 + || c == 5760 + || (8192..=8202).contains(&c) + || (8232..=8233).contains(&c) + || c == 8239 + || c == 8287 + || c == 12288 + || c == 65279 + }) else { + return false; + }; + let Some(rest) = re_take_fixed(rest, 3, |c: u32| (48..=57).contains(&c)) else { + return false; + }; + let Some(rest) = re_take_class(rest, 0, usize::MAX, |c: u32| { + (9..=13).contains(&c) + || c == 32 + || (45..=47).contains(&c) + || c == 160 + || c == 5760 + || (8192..=8202).contains(&c) + || (8232..=8233).contains(&c) + || c == 8239 + || c == 8287 + || c == 12288 + || c == 65279 + }) else { + return false; + }; + let Some(rest) = re_take_fixed(rest, 3, |c: u32| (48..=57).contains(&c)) else { + return false; + }; + let Some(rest) = re_take_class(rest, 0, usize::MAX, |c: u32| { + (9..=13).contains(&c) + || c == 32 + || (45..=47).contains(&c) + || c == 160 + || c == 5760 + || (8192..=8202).contains(&c) + || (8232..=8233).contains(&c) + || c == 8239 + || c == 8287 + || c == 12288 + || c == 65279 + }) else { + return false; + }; + let Some(rest) = re_take_fixed(rest, 4, |c: u32| (48..=57).contains(&c)) else { + return false; + }; + let Some(rest) = re_take_class(rest, 0, usize::MAX, |c: u32| { + (9..=13).contains(&c) + || c == 32 + || (45..=47).contains(&c) + || c == 160 + || c == 5760 + || (8192..=8202).contains(&c) + || (8232..=8233).contains(&c) + || c == 8239 + || c == 8287 + || c == 12288 + || c == 65279 + }) else { + return false; + }; + let Some(rest) = re_take_fixed(rest, 2, |c: u32| (48..=57).contains(&c)) else { + return false; + }; + rest.is_empty() +} +pub fn re_match_3(value: &str) -> bool { + let rest = value; + let Some(rest) = re_take_fixed(rest, 3, |c: u32| (48..=57).contains(&c)) else { + return false; + }; + let Some(rest) = re_take_class(rest, 0, usize::MAX, |c: u32| { + (9..=13).contains(&c) + || c == 32 + || (45..=47).contains(&c) + || c == 160 + || c == 5760 + || (8192..=8202).contains(&c) + || (8232..=8233).contains(&c) + || c == 8239 + || c == 8287 + || c == 12288 + || c == 65279 + }) else { + return false; + }; + let Some(rest) = re_take_fixed(rest, 3, |c: u32| (48..=57).contains(&c)) else { + return false; + }; + let Some(rest) = re_take_class(rest, 0, usize::MAX, |c: u32| { + (9..=13).contains(&c) + || c == 32 + || (45..=47).contains(&c) + || c == 160 + || c == 5760 + || (8192..=8202).contains(&c) + || (8232..=8233).contains(&c) + || c == 8239 + || c == 8287 + || c == 12288 + || c == 65279 + }) else { + return false; + }; + let Some(rest) = re_take_fixed(rest, 3, |c: u32| (48..=57).contains(&c)) else { + return false; + }; + let Some(rest) = re_take_class(rest, 0, usize::MAX, |c: u32| { + (9..=13).contains(&c) + || c == 32 + || (45..=47).contains(&c) + || c == 160 + || c == 5760 + || (8192..=8202).contains(&c) + || (8232..=8233).contains(&c) + || c == 8239 + || c == 8287 + || c == 12288 + || c == 65279 + }) else { + return false; + }; + let Some(rest) = re_take_fixed(rest, 2, |c: u32| (48..=57).contains(&c)) else { + return false; + }; + rest.is_empty() +} + +/// Runs each task on its own thread and answers the first one that lands `Some`. Cancellation is +/// best effort and semantically unobservable, exactly as `docs/semantics.md` describes: a losing +/// task may run to completion, and its answer is dropped on the floor. Tasks are matched to +/// completion order through a channel, not through joining threads in task order, which is what +/// makes this "the first task that answers" rather than "the first task in the list". +pub fn race_first_some<'scope, T: Send + 'scope>( + tasks: Vec Option + Send + 'scope>>, +) -> Option { + let count = tasks.len(); + let (sender, receiver) = std::sync::mpsc::channel::>(); + std::thread::scope(|scope| { + for task in tasks { + let sender = sender.clone(); + scope.spawn(move || { + let _ = sender.send(task()); + }); + } + drop(sender); + for _ in 0..count { + if let Ok(Some(value)) = receiver.recv() { + return Some(value); + } + } + None + }) +} + +/// One request or response header. Headers are an ordered list, never a map, so every target +/// preserves order and duplicates. +#[derive(Clone, Debug, PartialEq)] +pub struct HttpHeader { + pub name: String, + pub value: String, +} + +/// A request handed to the Http capability. The host adds no retries and no hidden headers. +#[derive(Clone, Debug, PartialEq)] +pub struct HttpRequest { + pub method: String, + pub url: String, + pub headers: Vec, + pub body: String, + pub timeout_millis: i64, +} + +/// A response from the Http capability. A status of 400 or more is a value, not a failure. +#[derive(Clone, Debug, PartialEq)] +pub struct HttpResponse { + pub status: i64, + pub headers: Vec, + pub body: String, +} + +/// Everything the generated core needs from the outside world. `std` has neither an HTTP client +/// nor a source of randomness or wall-clock time in one place, so the core takes this trait and +/// leaves the default implementation to the host — the same split every other capability +/// intrinsic uses, and the first friction `docs/targets/rust-sketch.md` predicted. +/// +/// `task.race` spawns one thread per task (see `race_first_some` above), so an implementation has +/// to be safe to share across threads; `Sync` is what that costs a hand-written implementation +/// that `std` alone cannot supply. +pub trait Capabilities: Sync { + /// A transport error or a timeout answers `None`; a 4xx or 5xx status is a value. + fn request(&self, request: HttpRequest) -> Option; + fn now(&self) -> i64; + fn sleep(&self, millis: i64); + fn next_u32(&self) -> i64; +} diff --git a/core/out/typescript/API.json b/core/out/typescript/API.json new file mode 100644 index 000000000..b4ca3dd40 --- /dev/null +++ b/core/out/typescript/API.json @@ -0,0 +1,233 @@ +{ + "functions": [ + { + "name": "formatCnpj", + "module": "format-cnpj.ts", + "params": [ + { + "name": "value", + "type": "string" + }, + { + "name": "options", + "type": "FormatCnpjOptions" + } + ], + "returns": "string", + "effects": [] + }, + { + "name": "formatCurrency", + "module": "format-currency.ts", + "params": [ + { + "name": "value", + "type": "number" + }, + { + "name": "symbol", + "type": "boolean" + } + ], + "returns": "string", + "effects": [] + }, + { + "name": "generateCnpj", + "module": "generate-cnpj.ts", + "params": [], + "returns": "string", + "effects": [] + }, + { + "name": "generateCpf", + "module": "generate-cpf.ts", + "params": [], + "returns": "string", + "effects": [] + }, + { + "name": "getAddressInfoByCep", + "module": "get-address-info-by-cep.ts", + "params": [ + { + "name": "cep", + "type": "string" + } + ], + "returns": "AddressInfo", + "effects": [ + "Fail", + "Fail" + ] + }, + { + "name": "getHolidays", + "module": "get-holidays.ts", + "params": [ + { + "name": "year", + "type": "number" + } + ], + "returns": "readonly Holiday[]", + "effects": [] + }, + { + "name": "isBusinessDay", + "module": "is-business-day.ts", + "params": [ + { + "name": "value", + "type": "number" + }, + { + "name": "includeOptional", + "type": "boolean" + } + ], + "returns": "boolean", + "effects": [] + }, + { + "name": "isValidCnpj", + "module": "is-valid-cnpj.ts", + "params": [ + { + "name": "cnpj", + "type": "string" + }, + { + "name": "version", + "type": "\"1\" | \"2\"" + } + ], + "returns": "boolean", + "effects": [] + }, + { + "name": "isValidCpf", + "module": "is-valid-cpf.ts", + "params": [ + { + "name": "cpf", + "type": "string" + } + ], + "returns": "boolean", + "effects": [] + } + ], + "seams": [ + { + "name": "generateCnpjWith", + "publicName": "generateCnpj", + "module": "generate-cnpj.ts", + "params": [ + { + "name": "env", + "type": "Capabilities" + } + ], + "returns": "string", + "hasWrapper": true + }, + { + "name": "generateCpfWith", + "publicName": "generateCpf", + "module": "generate-cpf.ts", + "params": [ + { + "name": "env", + "type": "Capabilities" + } + ], + "returns": "string", + "hasWrapper": true + }, + { + "name": "getAddressInfoByCepWith", + "publicName": "getAddressInfoByCep", + "module": "get-address-info-by-cep.ts", + "params": [ + { + "name": "cep", + "type": "string" + }, + { + "name": "env", + "type": "Capabilities" + } + ], + "returns": "AddressInfo", + "hasWrapper": true + } + ], + "records": [ + { + "name": "FormatCnpjOptions", + "fields": [ + { + "name": "pad", + "type": "boolean" + }, + { + "name": "version", + "type": "\"1\" | \"2\"" + }, + { + "name": "obfuscate", + "type": "boolean" + } + ] + }, + { + "name": "AddressInfo", + "fields": [ + { + "name": "cep", + "type": "string" + }, + { + "name": "state", + "type": "string" + }, + { + "name": "city", + "type": "string" + }, + { + "name": "neighborhood", + "type": "string" + }, + { + "name": "street", + "type": "string" + } + ] + }, + { + "name": "Holiday", + "fields": [ + { + "name": "name", + "type": "string" + }, + { + "name": "date", + "type": "number" + }, + { + "name": "type", + "type": "\"national\" | \"optional\" | \"religious\" | \"state\"" + } + ] + } + ], + "errors": [ + "GetAddressInfoByCepError", + "GetAddressInfoByCepNotFoundError", + "GetAddressInfoByCepValidationError", + "HttpError" + ] +} diff --git a/core/out/typescript/LOWERING.md b/core/out/typescript/LOWERING.md new file mode 100644 index 000000000..392fb3a9d --- /dev/null +++ b/core/out/typescript/LOWERING.md @@ -0,0 +1,309 @@ +# Lowering selections — typescript + +Generated by the engine. Each row is one operation, the argument types it was called +with, the implementation that was selected, and the rule that decided it. + +| operation | argument types | implementation | why | +| --------------------- | ---------------------------------------------------------------- | -------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `clock.sleep` | `Duration` | native | only candidate, cost none/constant | +| `core.eq` | `"1" \| "2", "2"` | native | only candidate, cost none/constant | +| `core.eq` | `"national" \| "optional" \| "religious" \| "state", "optional"` | native | only candidate, cost none/constant | +| `core.eq` | `Ascii[1], Ascii[1]` | native | only candidate, cost none/constant | +| `core.eq` | `Ascii[1], Digits[1]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[-1..1], Int[0..0]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[-2..2], Int[0..0]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[-48..79], Int[0..9]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[-1..-1]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[0..127]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[10..10]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[110..110]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[114..114]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[116..116]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[117..117]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[13..13]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[32..32]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[34..34]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[58..58]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[9..9]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..1114111], Int[92..92]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..2147483647], Int[11..11]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..2147483647], Int[14..14]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..3506328], Int[306..-1]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..9], Int[0..9]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..9600], Int[0..-1]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..7], Int[6..6]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[1..7], Int[7..7]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[48..57], Int[48..57]` | native | only candidate, cost none/constant | +| `core.eq` | `String[0..2147483647], Ascii[0]` | native | only candidate, cost none/constant | +| `date.addDays` | `CivilDate, Int[-2..-2]` | native | only candidate, cost none/constant | +| `date.addDays` | `CivilDate, Int[-47..-47]` | native | only candidate, cost none/constant | +| `date.addDays` | `CivilDate, Int[60..60]` | native | only candidate, cost none/constant | +| `date.clampEpochDays` | `Int[0..0]` | native | only candidate, cost none/constant | +| `date.compare` | `CivilDate, CivilDate` | native | only candidate, cost none/constant | +| `date.dayOfWeek` | `CivilDate` | native | only candidate, cost none/constant | +| `date.fromYmd` | `Int[1900..2099], Int[1..1], Int[1..1]` | portable | only candidate, cost none/constant | +| `date.fromYmd` | `Int[1900..2099], Int[10..10], Int[12..12]` | portable | only candidate, cost none/constant | +| `date.fromYmd` | `Int[1900..2099], Int[11..11], Int[15..15]` | portable | only candidate, cost none/constant | +| `date.fromYmd` | `Int[1900..2099], Int[11..11], Int[2..2]` | portable | only candidate, cost none/constant | +| `date.fromYmd` | `Int[1900..2099], Int[12..12], Int[25..25]` | portable | only candidate, cost none/constant | +| `date.fromYmd` | `Int[1900..2099], Int[3..3], Int[22..31]` | portable | only candidate, cost none/constant | +| `date.fromYmd` | `Int[1900..2099], Int[4..4], Int[1..25]` | portable | only candidate, cost none/constant | +| `date.fromYmd` | `Int[1900..2099], Int[4..4], Int[21..21]` | portable | only candidate, cost none/constant | +| `date.fromYmd` | `Int[1900..2099], Int[5..5], Int[1..1]` | portable | only candidate, cost none/constant | +| `date.fromYmd` | `Int[1900..2099], Int[9..9], Int[7..7]` | portable | only candidate, cost none/constant | +| `date.fromYmd` | `Int[2024..2099], Int[11..11], Int[20..20]` | portable | only candidate, cost none/constant | +| `date.year` | `CivilDate` | portable | only candidate, cost none/constant | +| `dec.abs` | `Decimal<2>` | native | only candidate, cost none/constant | +| `dec.isNegative` | `Decimal<2>` | native | only candidate, cost none/constant | +| `dec.unscaled` | `Decimal<2>` | native | only candidate, cost none/constant | +| `http.request` | `HttpRequest` | native | only candidate, cost many/linear | +| `int.add` | `Int[-10012..20013], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-146097..3506328], Int[-3506503..3798697]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-238871..9], Int[3..3]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-3504000..3795635], Int[-2400..2599]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-3506503..3798330], Int[0..367]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-3508380..3800745], Int[-2403..2603]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-3508623..3800862], Int[-95..103]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-36547330..36546740], Int[2..2]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-7..35], Int[114..114]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-719162..2932896], Int[719468..719468]` | native | only candidate, cost none/constant | +| `int.add` | `Int[-9612..10413], Int[-400..9600]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..1683], Int[2..2]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..17], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..18], Int[0..319]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..2147483646], Int[0..4]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..2147483646], Int[5..5]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..29], Int[0..6]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..30], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..337], Int[0..132]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..337], Int[1..31]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..342], Int[19..20]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..65520], Int[0..15]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..720], Int[0..90]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..891], Int[0..81]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..891], Int[0..99]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..9007199254740991], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..9007199254740991], Int[2..2]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..9007199254740991], Int[6..6]` | native | only candidate, cost none/constant | +| `int.add` | `Int[1..2], Int[9..9]` | native | only candidate, cost none/constant | +| `int.add` | `Int[1..31], Int[0..31]` | native | only candidate, cost none/constant | +| `int.add` | `Int[32..32], Int[0..6]` | native | only candidate, cost none/constant | +| `int.add` | `Int[32..38], Int[0..48]` | native | only candidate, cost none/constant | +| `int.add` | `Int[5..2147483658], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[5..2147483659], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[6..2147483667], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[6..2147483668], Int[1..1]` | native | only candidate, cost none/constant | +| `int.add` | `Int[8..352], Int[15..15]` | native | only candidate, cost none/constant | +| `int.add` | `Int[9..2147483671], Int[0..3]` | native | only candidate, cost none/constant | +| `int.div` | `Int[-3506022..3798461], Int[1460..1460]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.div` | `Int[-3506022..3798461], Int[146096..146096]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.div` | `Int[-3506022..3798461], Int[36524..36524]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.div` | `Int[-3508743..3800988], Int[365..365]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.div` | `Int[-36547328..36546742], Int[153..153]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.div` | `Int[-9600..10399], Int[100..100]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.div` | `Int[-9600..10399], Int[4..4]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.div` | `Int[-9612..10413], Int[100..100]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.div` | `Int[-9612..10413], Int[4..4]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.div` | `Int[0..469], Int[451..451]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.div` | `Int[0..99], Int[4..4]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.div` | `Int[0..9999], Int[400..400]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.div` | `Int[107..149], Int[31..31]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.div` | `Int[19..20], Int[4..4]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.div` | `Int[1900..2099], Int[100..100]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.div` | `Int[2..1685], Int[5..5]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.div` | `Int[306..3652364], Int[146097..146097]` | native | only candidate, cost none/constant; `/` on numbers is not integer division | +| `int.ge` | `Int[0..1114111], Int[0..0]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..1114111], Int[48..48]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..1114111], Int[65..65]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..1114111], Int[97..97]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..127], Int[65..65]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..17], Int[0..2147483647]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..599], Int[200..200]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[1900..2099], Int[2024..2024]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[0..14], Int[0..0]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[0..2], Int[0..0]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1..12], Int[2..2]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[1900..9999], Int[2099..2099]` | native | only candidate, cost none/constant | +| `int.le` | `Int[-238868..238858], Int[2..2]` | native | only candidate, cost none/constant | +| `int.le` | `Int[1..12], Int[2..2]` | native | only candidate, cost none/constant | +| `int.le` | `Int[22..56], Int[31..31]` | native | only candidate, cost none/constant | +| `int.le` | `Int[48..1114111], Int[57..57]` | native | only candidate, cost none/constant | +| `int.le` | `Int[65..1114111], Int[70..70]` | native | only candidate, cost none/constant | +| `int.le` | `Int[65..127], Int[90..90]` | native | only candidate, cost none/constant | +| `int.le` | `Int[97..1114111], Int[102..102]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[-238871..238867], Int[10..10]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[0..10], Int[2..2]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[0..17], Int[0..2147483647]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[0..4294967295], Int[4294967287..4294967296]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[0..9999], Int[0..0]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[1..9999], Int[1900..1900]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[200..599], Int[300..300]` | native | only candidate, cost none/constant | +| `int.lt` | `Int[306..3652364], Int[0..0]` | native | only candidate, cost none/constant | +| `int.max` | `Int[-10012..20014], Int[1..1]` | native | only candidate, cost none/constant | +| `int.max` | `Int[-4372068..6585557], Int[-719162..-719162]` | native | only candidate, cost none/constant | +| `int.max` | `Int[1..15], Int[0..0]` | native | only candidate, cost none/constant | +| `int.max` | `Int[1..62], Int[22..22]` | native | only candidate, cost none/constant | +| `int.min` | `Int[-719162..6585557], Int[2932896..2932896]` | native | only candidate, cost none/constant | +| `int.min` | `Int[0..65535], Int[65535..65535]` | native | only candidate, cost none/constant | +| `int.min` | `Int[1..20014], Int[9999..9999]` | native | only candidate, cost none/constant | +| `int.min` | `Int[22..62], Int[56..56]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[-14..14], Int[3..3]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[0..4294967295], Int[10..10]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[0..810], Int[11..11]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[0..86], Int[7..7]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[0..972], Int[11..11]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[0..99], Int[4..4]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[0..990], Int[11..11]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[107..149], Int[31..31]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[19..20], Int[4..4]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[1900..2099], Int[100..100]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[1900..2099], Int[19..19]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[23..367], Int[30..30]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[-1..24], Int[146097..146097]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[-1..24], Int[400..400]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[-9600..10399], Int[365..365]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..1], Int[31..31]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..24], Int[146097..146097]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..24], Int[400..400]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..4095], Int[16..16]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..9], Int[2..10]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..9], Int[2..11]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..9], Int[2..9]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[11..11], Int[0..29]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[153..153], Int[0..11]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[19..19], Int[0..18]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[2..2], Int[0..24]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[2..2], Int[0..3]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[22..22], Int[0..6]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[365..365], Int[-9612..10413]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[5..5], Int[-7309466..7309348]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[7..7], Int[0..1]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3506022..3798461], Int[-2401..2601]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3506022..3798461], Int[-3510887..3803444]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3506400..3798234], Int[-96..103]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3508718..3800965], Int[-23..25]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3510783..3803348], Int[-96..104]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[-3652600..7305025], Int[719468..719468]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..127], Int[48..48]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..15], Int[1..14]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..24], Int[1..1]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..35], Int[0..7]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[0..9999], Int[-400..9600]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[1..368], Int[1..1]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[1..9999], Int[1..1]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[10..10], Int[0..8]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[10..238867], Int[9..9]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[11..11], Int[0..9]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[11..11], Int[2..10]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[14..358], Int[6..6]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[19..362], Int[4..5]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[3..12], Int[3..3]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[3..17], Int[2..2]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[3..4], Int[3..3]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[3..86], Int[0..3]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[306..3652364], Int[-146097..3506328]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[32..56], Int[31..31]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[32..86], Int[0..29]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[48..57], Int[48..48]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[65..70], Int[55..55]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[97..102], Int[87..87]` | native | only candidate, cost none/constant | +| `opt.isNone` | `Option` | native | only candidate, cost none/constant | +| `opt.isNone` | `Option` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Ascii[0]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, CivilDate` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[-1..-1]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[0..0]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[48..48]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[-2..-2]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[0..0]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[48..48]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Ascii[0]` | native | only candidate, cost none/constant | +| `opt.unwrap` | `Option` | native | only candidate, cost none/constant | +| `opt.unwrap` | `Option` | native | only candidate, cost none/constant | +| `random.nextU32` | `` | native | only candidate, cost none/constant | +| `re.retain` | `Ascii[1..15]` | native | only candidate, cost one/linear; a single pass with the negated class, which every engine reads the same way | +| `re.retain` | `String[0..2147483647]` | native | only candidate, cost one/linear; a single pass with the negated class, which every engine reads the same way | +| `re.test` | `String[0..2147483647]` | native | only candidate, cost none/linear; the normalized pattern is inside the compatibility subset | +| `seq.at` | `List[0..2147483647], Int[0..2147483650]` | native | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[0..9007199254740991]` | native | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[1..9007199254740992]` | native | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[5..2147483658]` | native | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[5..2147483659]` | native | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[6..2147483667]` | native | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[6..2147483668]` | native | only candidate, cost none/constant | +| `seq.at` | `List[0..2147483647], Int[9..2147483674]` | native | only candidate, cost none/constant | +| `seq.at` | `List[5..5], Int[0..4]` | native | only candidate, cost none/constant | +| `seq.at` | `List[0..15], Int[0..14]` | native | only candidate, cost none/constant | +| `seq.get` | `List[12..12], Int[0..11]` | native | only candidate, cost none/constant | +| `seq.len` | `List[0..2147483647]` | native | only candidate, cost none/constant | +| `seq.len` | `List[5..5]` | native | only candidate, cost none/constant | +| `seq.len` | `List[12..12]` | native | only candidate, cost none/constant | +| `seq.len` | `List[0..15]` | native | only candidate, cost none/constant | +| `seq.push` | `` | native | only candidate, cost none/constant | +| `seq.sortStableBy` | `List[12..13], (Holiday) => CivilDate` | native | only candidate, cost one/nlogn | +| `str.asciiUpper` | `Ascii[0..2147483647]` | native | only candidate, cost one/linear; `toUpperCase` is only ASCII-equivalent on ASCII input (ß becomes SS otherwise) | +| `str.asciiUpper` | `String[0..2147483647]` | native | native, cost one/linear; a single regex pass maps a-z and leaves every other scalar alone, which is the Core's rule for any input; rejected `toUpperCase` is only ASCII-equivalent on ASCII input (ß becomes SS otherwise) | +| `str.charAtOpt` | `Ascii[0..2147483647], Int[0..17]` | native | only candidate, cost none/constant; indexing a string yields one UTF-16 code unit, and undefined past the end | +| `str.charAtOpt` | `Ascii[18], Int[0..17]` | native | only candidate, cost none/constant; indexing a string yields one UTF-16 code unit, and undefined past the end | +| `str.codeAt` | `Ascii[14], Int[12..12]` | native | only candidate, cost none/constant; `charCodeAt` returns a UTF-16 code unit | +| `str.codeAt` | `Ascii[14], Int[13..13]` | native | only candidate, cost none/constant; `charCodeAt` returns a UTF-16 code unit | +| `str.codeAt` | `Digits[11], Int[0..0]` | native | only candidate, cost none/constant; `charCodeAt` returns a UTF-16 code unit | +| `str.codeAt` | `Digits[11], Int[0..8]` | native | only candidate, cost none/constant; `charCodeAt` returns a UTF-16 code unit | +| `str.codeAt` | `Digits[11], Int[0..9]` | native | only candidate, cost none/constant; `charCodeAt` returns a UTF-16 code unit | +| `str.codeAt` | `Digits[11], Int[1..10]` | native | only candidate, cost none/constant; `charCodeAt` returns a UTF-16 code unit | +| `str.codeAt` | `Digits[11], Int[10..10]` | native | only candidate, cost none/constant; `charCodeAt` returns a UTF-16 code unit | +| `str.codeAt` | `Digits[11], Int[9..9]` | native | only candidate, cost none/constant; `charCodeAt` returns a UTF-16 code unit | +| `str.codeAt` | `Digits[12], Int[0..0]` | native | only candidate, cost none/constant; `charCodeAt` returns a UTF-16 code unit | +| `str.codeAt` | `Digits[12], Int[1..11]` | native | only candidate, cost none/constant; `charCodeAt` returns a UTF-16 code unit | +| `str.codeAt` | `Digits[14], Int[0..0]` | native | only candidate, cost none/constant; `charCodeAt` returns a UTF-16 code unit | +| `str.codeAt` | `Digits[14], Int[0..11]` | native | only candidate, cost none/constant; `charCodeAt` returns a UTF-16 code unit | +| `str.codeAt` | `Digits[14], Int[1..13]` | native | only candidate, cost none/constant; `charCodeAt` returns a UTF-16 code unit | +| `str.codeAtOpt` | `Ascii[0..2147483647], Int[0..2147483646]` | native | only candidate, cost none/constant; charCodeAt answers NaN past the end, so the bound is checked explicitly | +| `str.codePoints` | `Ascii[5]` | native | only candidate, cost one/linear | +| `str.codePoints` | `Digits[0..15]` | native | only candidate, cost one/linear | +| `str.codePoints` | `String[0..2147483647]` | native | only candidate, cost one/linear | +| `str.concat` | `Ascii[0..17], Ascii[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Ascii[0..3], Ascii[1..47]` | native | only candidate, cost none/constant | +| `str.concat` | `Ascii[0..30], Ascii[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Ascii[1..31], Ascii[0..16]` | native | only candidate, cost none/constant | +| `str.concat` | `Ascii[1..4], Ascii[1..47]` | native | only candidate, cost none/constant | +| `str.concat` | `Ascii[1], Ascii[0..3]` | native | only candidate, cost none/constant | +| `str.concat` | `Ascii[1], Ascii[3]` | native | only candidate, cost none/constant | +| `str.concat` | `Ascii[2], Ascii[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Ascii[25], Digits[8] matches ^[0-9]{8}$` | native | only candidate, cost none/constant | +| `str.concat` | `Ascii[33], Ascii[6]` | native | only candidate, cost none/constant | +| `str.concat` | `Ascii[36], Digits[8] matches ^[0-9]{8}$` | native | only candidate, cost none/constant | +| `str.concat` | `Ascii[4], Ascii[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Digits[1], Digits[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Digits[10], Digits[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Digits[11], Digits[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Digits[12], Digits[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Digits[12], Digits[2]` | native | only candidate, cost none/constant | +| `str.concat` | `Digits[13], Digits[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Digits[2], Digits[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Digits[3], Digits[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Digits[4], Digits[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Digits[5], Digits[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Digits[6], Digits[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Digits[7], Digits[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Digits[8], Digits[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Digits[9], Digits[1]` | native | only candidate, cost none/constant | +| `str.concat` | `Digits[9], Digits[2]` | native | only candidate, cost none/constant | +| `str.fromCodePoints` | `List[0..2147483647]` | native | only candidate, cost one/linear | +| `str.fromCodePoints` | `List[0..30]` | native | only candidate, cost one/linear | +| `str.fromCodePoints` | `List[1..1]` | native | only candidate, cost one/linear | +| `str.fromInt` | `Int[-9007199254740991..9007199254740991]` | native | only candidate, cost one/linear | +| `str.fromInt` | `Int[0..9]` | native | only candidate, cost one/linear | +| `str.len` | `Ascii[0..2147483647]` | native | only candidate, cost none/constant; `String#length` counts UTF-16 code units, which only equals the scalar count for ASCII | +| `str.len` | `Ascii[18]` | native | only candidate, cost none/constant; `String#length` counts UTF-16 code units, which only equals the scalar count for ASCII | +| `str.len` | `Ascii[3..17]` | native | only candidate, cost none/constant; `String#length` counts UTF-16 code units, which only equals the scalar count for ASCII | +| `str.len` | `Digits[0..2147483647]` | native | only candidate, cost none/constant; `String#length` counts UTF-16 code units, which only equals the scalar count for ASCII | +| `str.len` | `Digits[12]` | native | only candidate, cost none/constant; `String#length` counts UTF-16 code units, which only equals the scalar count for ASCII | +| `str.padStart` | `Ascii[0..2147483647], Int[0..18], Digits[1]` | native | only candidate, cost one/linear; `padStart` counts UTF-16 code units | +| `str.padStart` | `Ascii[1..17], Int[3..3], Digits[1]` | native | only candidate, cost one/linear; `padStart` counts UTF-16 code units | +| `str.slice` | `Ascii[3..17], Int[0..0], Int[1..15]` | native | only candidate, cost one/linear; `slice` cuts at UTF-16 boundaries | +| `str.slice` | `Ascii[3..17], Int[1..15], Int[3..17]` | native | only candidate, cost one/linear; `slice` cuts at UTF-16 boundaries | +| `str.trim` | `String[0..2147483647]` | native | only candidate, cost one/linear; `String#trim` removes exactly the 25 code points the spec names | +| `task.race` | `List<() => Option>[2..2]` | native | only candidate, cost many/linear; Promise.any resolves with the first task to answer, which is the semantics of race | + +Mix: 288 native, 0 library, 12 portable. diff --git a/core/out/typescript/SIZE.json b/core/out/typescript/SIZE.json new file mode 100644 index 000000000..4786e590a --- /dev/null +++ b/core/out/typescript/SIZE.json @@ -0,0 +1,54 @@ +{ + "exports": { + "formatCnpj": { + "minified": 518, + "gzip": 334, + "brotli": 292 + }, + "formatCurrency": { + "minified": 434, + "gzip": 312, + "brotli": 267 + }, + "generateCnpj": { + "minified": 1301, + "gzip": 629, + "brotli": 549 + }, + "generateCpf": { + "minified": 1284, + "gzip": 614, + "brotli": 532 + }, + "getAddressInfoByCep": { + "minified": 2855, + "gzip": 1277, + "brotli": 1149 + }, + "getHolidays": { + "minified": 2379, + "gzip": 817, + "brotli": 765 + }, + "isBusinessDay": { + "minified": 2922, + "gzip": 1022, + "brotli": 966 + }, + "isValidCnpj": { + "minified": 1481, + "gzip": 499, + "brotli": 452 + }, + "isValidCpf": { + "minified": 760, + "gzip": 333, + "brotli": 290 + } + }, + "total": { + "minified": 9839, + "gzip": 3175, + "brotli": 2891 + } +} diff --git a/core/out/typescript/SOURCEMAP.json b/core/out/typescript/SOURCEMAP.json new file mode 100644 index 000000000..f93c1b3f2 --- /dev/null +++ b/core/out/typescript/SOURCEMAP.json @@ -0,0 +1,197 @@ +{ + "lib/format.ts#patternSlots": { + "module": "lib/format", + "start": 1345, + "end": 2086 + }, + "lib/format.ts#formatWithPattern": { + "module": "lib/format", + "start": 2182, + "end": 2821 + }, + "format-cnpj.ts#formatCnpj": { + "module": "format-cnpj", + "start": 967, + "end": 1233 + }, + "format-currency.ts#formatCurrency": { + "module": "format-currency", + "start": 966, + "end": 1499 + }, + "lib/random.ts#randomBelow": { + "module": "lib/random", + "start": 1137, + "end": 1640 + }, + "lib/cnpj.ts#randomCnpjBase": { + "module": "lib/cnpj", + "start": 2700, + "end": 2947 + }, + "lib/cnpj.ts#cnpjCheckDigit": { + "module": "lib/cnpj", + "start": 687, + "end": 1184 + }, + "lib/cnpj.ts#hasLetter": { + "module": "lib/cnpj", + "start": 1706, + "end": 2363 + }, + "lib/cnpj.ts#hasValidCnpjChecksum": { + "module": "lib/cnpj", + "start": 1265, + "end": 1478 + }, + "lib/cnpj.ts#isRepeatedCnpj": { + "module": "lib/cnpj", + "start": 3028, + "end": 3247 + }, + "lib/digits.ts#isRepeatedRun": { + "module": "lib/digits", + "start": 1135, + "end": 1358 + }, + "generate-cnpj.ts#generateCnpj": { + "module": "generate-cnpj", + "start": 982, + "end": 1571 + }, + "generate-cnpj.ts#generateCnpjWith": { + "module": "generate-cnpj", + "start": 982, + "end": 1571 + }, + "lib/cpf.ts#randomCpfBase": { + "module": "lib/cpf", + "start": 999, + "end": 1196 + }, + "lib/cpf.ts#cpfCheckDigit": { + "module": "lib/cpf", + "start": 394, + "end": 687 + }, + "lib/cpf.ts#cpfCheckDigit1": { + "module": "lib/cpf", + "start": 394, + "end": 687 + }, + "lib/cpf.ts#isRepeated": { + "module": "lib/cpf", + "start": 1283, + "end": 1499 + }, + "generate-cpf.ts#generateCpf": { + "module": "generate-cpf", + "start": 1014, + "end": 1566 + }, + "generate-cpf.ts#generateCpfWith": { + "module": "generate-cpf", + "start": 1014, + "end": 1566 + }, + "get-address-info-by-cep.ts#getWithRetry": { + "module": "get-address-info-by-cep", + "start": 1086, + "end": 1491 + }, + "get-address-info-by-cep.ts#isOk": { + "module": "get-address-info-by-cep", + "start": 1529, + "end": 1622 + }, + "get-address-info-by-cep.ts#fetchViaCep": { + "module": "get-address-info-by-cep", + "start": 1701, + "end": 2300 + }, + "get-address-info-by-cep.ts#fetchBrasilApi": { + "module": "get-address-info-by-cep", + "start": 2351, + "end": 2957 + }, + "get-address-info-by-cep.ts#getAddressInfoByCep": { + "module": "get-address-info-by-cep", + "start": 3382, + "end": 3789 + }, + "get-address-info-by-cep.ts#getAddressInfoByCepWith": { + "module": "get-address-info-by-cep", + "start": 3382, + "end": 3789 + }, + "lib/json.ts#matchesAt": { + "module": "lib/json", + "start": 582, + "end": 1320 + }, + "lib/json.ts#isSpace": { + "module": "lib/json", + "start": 1370, + "end": 1497 + }, + "lib/json.ts#hexValue": { + "module": "lib/json", + "start": 1568, + "end": 2072 + }, + "lib/json.ts#jsonStringField": { + "module": "lib/json", + "start": 2300, + "end": 4239 + }, + "lib/easter.ts#easterDayOfMarch": { + "module": "lib/easter", + "start": 527, + "end": 1128 + }, + "lib/civil.ts#civilDate10": { + "module": "lib/civil", + "start": 445, + "end": 624 + }, + "get-holidays.ts#getHolidays": { + "module": "get-holidays", + "start": 898, + "end": 2426 + }, + "is-business-day.ts#isBusinessDay": { + "module": "is-business-day", + "start": 578, + "end": 1064 + }, + "is-valid-cnpj.ts#isValidCnpj": { + "module": "is-valid-cnpj", + "start": 1397, + "end": 2445 + }, + "is-valid-cpf.ts#isValidCpf": { + "module": "is-valid-cpf", + "start": 793, + "end": 1143 + }, + "std/date.ts#floorDiv1": { + "module": "std/date", + "start": 432, + "end": 655 + }, + "std/date.ts#daysFromCivil": { + "module": "std/date", + "start": 752, + "end": 1560 + }, + "std/date.ts#floorDiv": { + "module": "std/date", + "start": 432, + "end": 655 + }, + "std/date.ts#yearFromDays": { + "module": "std/date", + "start": 1627, + "end": 2212 + } +} diff --git a/core/out/typescript/_driver.ts b/core/out/typescript/_driver.ts new file mode 100644 index 000000000..0794a9097 --- /dev/null +++ b/core/out/typescript/_driver.ts @@ -0,0 +1,150 @@ +// Code generated by the logic engine. DO NOT EDIT. +// source: _driver + +import { createInterface } from "node:readline"; +import { formatCnpj } from "./format-cnpj.ts"; +import { formatCurrency } from "./format-currency.ts"; +import { generateCnpjWith } from "./generate-cnpj.ts"; +import { generateCpfWith } from "./generate-cpf.ts"; +import { getAddressInfoByCepWith } from "./get-address-info-by-cep.ts"; +import { getHolidays } from "./get-holidays.ts"; +import { isBusinessDay } from "./is-business-day.ts"; +import { isValidCnpj } from "./is-valid-cnpj.ts"; +import { isValidCpf } from "./is-valid-cpf.ts"; +import { readFileSync, existsSync } from "node:fs"; +import { + defaultCapabilities, + type Capabilities, + type HttpRequest, + type HttpResponse, +} from "./capabilities.ts"; + +/** + * The reference PCG32: same constants and default seed as `Interpreter`'s, so a draw + * matches the reference bit for bit. A fresh instance is built for every request, the same + * way the reference model starts a fresh interpreter — and so a fresh generator — per case. + */ +class Pcg32 { + private state = 0n; + private readonly increment = 1442695040888963407n; + + constructor(seed: bigint) { + this.next(); + this.state = (this.state + seed) & 0xffffffffffffffffn; + this.next(); + } + + next(): number { + const previous = this.state; + this.state = + (previous * 6364136223846793005n + this.increment) & 0xffffffffffffffffn; + const xorshifted = (((previous >> 18n) ^ previous) >> 27n) & 0xffffffffn; + const rotation = previous >> 59n; + return Number( + ((xorshifted >> rotation) | (xorshifted << (-rotation & 31n))) & + 0xffffffffn, + ); + } +} + +// The interpreter's own default: its constructor falls back to this seed whenever +// `Capabilities.seed` is left unset, which is how every conformance case runs it. +const DEFAULT_SEED = 0x853c49e6748fea9bn; + +/** + * The capability fake the differential harness drives. + * + * Responses come from `fixtures.json`, a URL that is not in it models a transport error, + * and the scripted latency is what decides a race, the same way the reference model's + * virtual clock decides it. `nextU32` gets a fresh PCG32 per call, matching the reference + * model's fresh interpreter per case. + */ +type Fixture = { status: number; body: string; latencyMillis?: number }; + +function fakeCapabilities(fixtures: Record): Capabilities { + const random = new Pcg32(DEFAULT_SEED); + + return { + async request(request: HttpRequest): Promise { + const fixture = fixtures[request.url]; + + if (fixture === undefined) return undefined; + + await new Promise((resolve) => + setTimeout(resolve, fixture.latencyMillis ?? 0), + ); + + return { status: fixture.status, headers: [], body: fixture.body }; + }, + now: () => 0, + sleep: (milliseconds: number) => + new Promise((resolve) => setTimeout(resolve, milliseconds)), + nextU32: () => random.next(), + }; +} + +type Handler = (args: readonly unknown[], env: unknown) => unknown; + +const handlers: Record = { + "format-cnpj::formatCnpj": (args, env) => + formatCnpj( + args[0] as Parameters[0], + args[1] as Parameters[1], + ), + "format-currency::formatCurrency": (args, env) => + formatCurrency( + args[0] as Parameters[0], + args[1] as Parameters[1], + ), + "generate-cnpj::generateCnpj": (args, env) => + generateCnpjWith(env as Capabilities), + "generate-cpf::generateCpf": (args, env) => + generateCpfWith(env as Capabilities), + "get-address-info-by-cep::getAddressInfoByCep": (args, env) => + getAddressInfoByCepWith( + args[0] as Parameters[0], + env as Capabilities, + ), + "get-holidays::getHolidays": (args, env) => + getHolidays(args[0] as Parameters[0]), + "is-business-day::isBusinessDay": (args, env) => + isBusinessDay( + args[0] as Parameters[0], + args[1] as Parameters[1], + ), + "is-valid-cnpj::isValidCnpj": (args, env) => + isValidCnpj( + args[0] as Parameters[0], + args[1] as Parameters[1], + ), + "is-valid-cpf::isValidCpf": (args, env) => + isValidCpf(args[0] as Parameters[0]), +}; + +const fixturePath = new URL("fixtures.json", import.meta.url).pathname; +const fixtures = existsSync(fixturePath) + ? (JSON.parse(readFileSync(fixturePath, "utf8")) as Record) + : undefined; + +const reader = createInterface({ input: process.stdin }); + +for await (const line of reader) { + if (line.trim() === "") continue; + const request = JSON.parse(line) as { fn: string; args: unknown[] }; + + // A fresh environment per line: nextU32 starts from the same state the reference model's + // fresh interpreter starts from for every case. + const environment = + fixtures === undefined ? defaultCapabilities() : fakeCapabilities(fixtures); + + try { + const value = await handlers[request.fn]!(request.args, environment); + process.stdout.write( + `${JSON.stringify({ ok: true, value: value === undefined ? null : value })}\n`, + ); + } catch (error) { + process.stdout.write( + `${JSON.stringify({ ok: false, error: (error as Error).constructor.name })}\n`, + ); + } +} diff --git a/core/out/typescript/capabilities.ts b/core/out/typescript/capabilities.ts new file mode 100644 index 000000000..1bfa0bf1a --- /dev/null +++ b/core/out/typescript/capabilities.ts @@ -0,0 +1,112 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: capabilities + +/** One request or response header. */ +export type HttpHeader = { + readonly name: string; + readonly value: string; +}; + +/** An Http request handed to the environment. */ +export type HttpRequest = { + readonly method: string; + readonly url: string; + readonly headers: readonly HttpHeader[]; + readonly body: string; + readonly timeoutMillis: number; +}; + +/** An Http response. A status of 400 or more is a value, not a failure. */ +export type HttpResponse = { + readonly status: number; + readonly headers: readonly HttpHeader[]; + readonly body: string; +}; + +/** Everything the core needs from the outside world. */ +export type Capabilities = { + /** A transport error or a timeout answers `undefined`; a 4xx or 5xx status is a value. */ + request(request: HttpRequest): Promise; + now(): number; + sleep(milliseconds: number): Promise; + nextU32(): number; +}; + +/** The default environment, built from the platform's own standard library. */ +export function defaultCapabilities(): Capabilities { + return { + async request(request: HttpRequest): Promise { + const controller = new AbortController(); + const timer = setTimeout(() => controller.abort(), request.timeoutMillis); + + try { + const response = await fetch(request.url, { + method: request.method, + headers: request.headers.map((header) => [header.name, header.value]), + body: request.body === "" ? undefined : request.body, + signal: controller.signal, + }); + + const headers: { name: string; value: string }[] = []; + response.headers.forEach((value, name) => + headers.push({ name, value }), + ); + + return { + status: response.status, + headers, + body: await response.text(), + }; + } catch { + return undefined; + } finally { + clearTimeout(timer); + } + }, + now(): number { + return Date.now(); + }, + sleep(milliseconds: number): Promise { + return new Promise((resolve) => setTimeout(resolve, milliseconds)); + }, + // Not cryptographically secure, deliberately: the utilities that draw are generating + // example documents, the published package documents using `Math.random()` for exactly + // that, and `crypto.getRandomValues` measures 130x the cost per draw. A caller who needs + // unpredictability passes its own capability. + nextU32(): number { + return Math.floor(Math.random() * 4294967296); + }, + }; +} + +/** + * The platform default, built once at module load rather than per call — every public wrapper + * (`docs/decisions/0011-public-entry-points-vs-capabilities.md`) shares this one instance, the + * same way a caller who builds their own environment would share it across calls. + */ +export const DEFAULT_CAPABILITIES: Capabilities = defaultCapabilities(); + +/** + * Takes the first task to answer, discarding the losers, and answers undefined when none does. + * + * Cancellation is best effort and semantically unobservable: a losing task may keep running, and + * its answer is dropped. Only idempotent work belongs inside a race. + */ +export async function raceFirstSome( + tasks: readonly (() => Promise)[], +): Promise { + try { + return await Promise.any( + tasks.map(async (task) => { + const value = await task(); + + if (value === undefined) throw new Error("no answer"); + + return value; + }), + ); + } catch { + return undefined; + } +} diff --git a/core/out/typescript/errors.ts b/core/out/typescript/errors.ts new file mode 100644 index 000000000..2f2942b92 --- /dev/null +++ b/core/out/typescript/errors.ts @@ -0,0 +1,18 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: errors + +/** The root of every domain error the core raises. */ +export class DomainError extends Error {} + +/** Raised by the engine's own intrinsics. */ +export class HttpError extends DomainError {} + +/** Base of every error this utility raises. */ +export class GetAddressInfoByCepError extends DomainError {} + +/** The value given is not a CEP. */ +export class GetAddressInfoByCepValidationError extends GetAddressInfoByCepError {} + +/** No CEP service knows this CEP, or none answered. */ +export class GetAddressInfoByCepNotFoundError extends GetAddressInfoByCepError {} diff --git a/core/out/typescript/fixtures.json b/core/out/typescript/fixtures.json new file mode 100644 index 000000000..f05926010 --- /dev/null +++ b/core/out/typescript/fixtures.json @@ -0,0 +1,27 @@ +{ + "https://viacep.com.br/ws/01310100/json/": { + "status": 200, + "body": "{\"cep\":\"01310-100\",\"logradouro\":\"Avenida Paulista\",\"bairro\":\"Bela Vista\",\"localidade\":\"São Paulo\",\"uf\":\"SP\"}", + "latencyMillis": 60 + }, + "https://brasilapi.com.br/api/cep/v1/01310100": { + "status": 200, + "body": "{\"cep\":\"01310100\",\"state\":\"SP\",\"city\":\"São Paulo\",\"neighborhood\":\"Bela Vista\",\"street\":\"Avenida Paulista\"}", + "latencyMillis": 20 + }, + "https://brasilapi.com.br/api/cep/v1/30130010": { + "status": 200, + "body": "{\"cep\":\"30130010\",\"state\":\"MG\",\"city\":\"Belo Horizonte\",\"neighborhood\":\"Centro\",\"street\":\"Avenida Afonso Pena\"}", + "latencyMillis": 40 + }, + "https://viacep.com.br/ws/99999999/json/": { + "status": 200, + "body": "{\"erro\":true}", + "latencyMillis": 10 + }, + "https://brasilapi.com.br/api/cep/v1/99999999": { + "status": 404, + "body": "{\"message\":\"not found\"}", + "latencyMillis": 10 + } +} diff --git a/core/out/typescript/format-cnpj.ts b/core/out/typescript/format-cnpj.ts new file mode 100644 index 000000000..8663b1ac1 --- /dev/null +++ b/core/out/typescript/format-cnpj.ts @@ -0,0 +1,32 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: format-cnpj +// content: 7541034a1f26 +import { formatWithPattern } from "./lib/format.ts"; + +export type FormatCnpjOptions = { + /** Whether to left pad the value with zeros up to the number of slots in the pattern. */ + readonly pad: boolean; + /** Which CNPJ format to read. */ + readonly version: "1" | "2"; + /** Whether to hide the first two characters and the two check digits with `*`. */ + readonly obfuscate: boolean; +}; + +/** + * Formats a CNPJ value as `00.000.000/0000-00`. + * + * The core takes a string and a fully normalized options record; reading a number, a missing + * options object or a truthy non-boolean is the DX's job. + */ +export function formatCnpj(value: string, options: FormatCnpjOptions): string { + const sanitized: string = + options.version === "2" + ? value.replace(/[^0-9A-Za-z]/gu, "").toUpperCase() + : value.replace(/[^0-9]/gu, ""); + return formatWithPattern( + sanitized, + options.obfuscate ? "**.000.000/0000-**" : "00.000.000/0000-00", + options.pad, + ); +} diff --git a/core/out/typescript/format-currency.ts b/core/out/typescript/format-currency.ts new file mode 100644 index 000000000..585db0f78 --- /dev/null +++ b/core/out/typescript/format-currency.ts @@ -0,0 +1,42 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: format-currency +// content: 1e0947bdf328 +/** + * Formats an exact amount in Brazilian Real, with two decimal places. + * + * The separators are the ones Lei nº 9.069/1995 art. 1º prescribes and the CLDR pt-BR data uses: + * `.` between thousands, `,` before the centavos, and a non-breaking space after `R$`. A negative + * amount puts the sign before the symbol, `-R$ 10,50`, the shape `Intl.NumberFormat` produces. + * + * Turning a host value into an exact amount is the DX's job, and so is the rounding that + * conversion needs; see docs/contracts.md, which records exactly how the published package rounds. + */ +export function formatCurrency(value: number, symbol: boolean): string { + const negative: boolean = value < 0; + const unscaled: number = Math.abs(value); + const digits: string = unscaled.toString().padStart(3, "0"); + const cut: number = Math.max(digits.length - 2, 0); + const whole: string = digits.slice(0, cut); + const cents: string = digits.slice(cut, digits.length); + const _inl26Whole: string = whole.replace(/[^0-9]/gu, ""); + let _inl23Out: number[] = []; + const _inl24Scalars: readonly number[] = Array.from( + _inl26Whole, + (scalar) => scalar.codePointAt(0)!, + ); + for (let _inl25Index = 0; _inl25Index < _inl24Scalars.length; _inl25Index++) { + if (_inl25Index > 0 && (_inl24Scalars.length - _inl25Index) % 3 === 0) { + _inl23Out.push(46); + } + _inl23Out.push(_inl24Scalars[_inl25Index] ?? 48); + } + const _inl27Result: string = _inl23Out + .map((point) => String.fromCodePoint(point)) + .join(""); + const body: string = _inl27Result + "," + cents; + const prefix: string = symbol + ? "R$" + [32].map((point) => String.fromCodePoint(point)).join("") + : ""; + return negative ? "-" + prefix + body : prefix + body; +} diff --git a/core/out/typescript/generate-cnpj.ts b/core/out/typescript/generate-cnpj.ts new file mode 100644 index 000000000..84f0e9129 --- /dev/null +++ b/core/out/typescript/generate-cnpj.ts @@ -0,0 +1,55 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: generate-cnpj +// content: 41b441a54f18 +import { cnpjCheckDigit, randomCnpjBase } from "./lib/cnpj.ts"; +import { isRepeatedRun } from "./lib/digits.ts"; +import type { Capabilities } from "./capabilities.ts"; +import { DEFAULT_CAPABILITIES, raceFirstSome } from "./capabilities.ts"; + +const generateCnpjTable1: readonly number[] = [ + 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2, +]; + +const generateCnpjTable2: readonly number[] = [ + 6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2, +]; + +/** + * Generates a valid random CNPJ (Cadastro Nacional da Pessoa Jurídica) in the numeric format: 14 + * digits, under the check digit rule both CNPJ versions share. + * + * Matches the published `generateCnpj()` called with no options: a random 8-digit root and + * 4-digit branch (the "número de ordem"), redrawn while every digit of the 12-digit base is the + * same, followed by its two check digits. The alphanumeric version and a chosen branch are DX + * concerns layered on the same base and check digit rule, not a different generator. + */ +export function generateCnpj(): string { + return generateCnpjWith(DEFAULT_CAPABILITIES); +} + +/** + * `generateCnpj`, taking its capabilities explicitly. + * + * The public `generateCnpj` calls this with the platform's defaults. Pass your own to + * supply a clock, a source of randomness or an HTTP client — which is what the + * differential conformance driver does to make a run reproducible. + */ +export function generateCnpjWith(env: Capabilities): string { + let base: string = randomCnpjBase(env); + for (let attempt = 0; attempt < 8; attempt++) { + if (!isRepeatedRun(base)) { + break; + } + base = randomCnpjBase(env); + } + const firstDigit: string = cnpjCheckDigit( + base + "00", + generateCnpjTable1, + ).toString(); + const secondDigit: string = cnpjCheckDigit( + base + firstDigit + "0", + generateCnpjTable2, + ).toString(); + return base + firstDigit + secondDigit; +} diff --git a/core/out/typescript/generate-cpf.ts b/core/out/typescript/generate-cpf.ts new file mode 100644 index 000000000..699f8a750 --- /dev/null +++ b/core/out/typescript/generate-cpf.ts @@ -0,0 +1,44 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: generate-cpf +// content: 00fc978e8350 +import { cpfCheckDigit, cpfCheckDigit1, randomCpfBase } from "./lib/cpf.ts"; +import { isRepeatedRun } from "./lib/digits.ts"; +import type { Capabilities } from "./capabilities.ts"; +import { DEFAULT_CAPABILITIES, raceFirstSome } from "./capabilities.ts"; + +/** + * Generates a valid random CPF (Cadastro de Pessoas Físicas): 11 digits, under the check digit + * rule (weights 10..2 and 11..2) the Receita Federal's Manual de Preenchimento da e-Financeira, + * Anexo II specifies. + * + * Matches the published `generateCpf()` called with no state: a random 9-digit base — 8 digits + * plus a região fiscal digit, also drawn at random here — redrawn while every digit of it is the + * same, followed by its two check digits. The state code option is a DX concern: it only ever + * picks which digit the 9th position draws from, never how the rest of the document is built. + */ +export function generateCpf(): string { + return generateCpfWith(DEFAULT_CAPABILITIES); +} + +/** + * `generateCpf`, taking its capabilities explicitly. + * + * The public `generateCpf` calls this with the platform's defaults. Pass your own to + * supply a clock, a source of randomness or an HTTP client — which is what the + * differential conformance driver does to make a run reproducible. + */ +export function generateCpfWith(env: Capabilities): string { + let base: string = randomCpfBase(env); + for (let attempt = 0; attempt < 8; attempt++) { + if (!isRepeatedRun(base)) { + break; + } + base = randomCpfBase(env); + } + const firstDigit: string = cpfCheckDigit(base + "00").toString(); + const secondDigit: string = cpfCheckDigit1( + base + firstDigit + "0", + ).toString(); + return base + firstDigit + secondDigit; +} diff --git a/core/out/typescript/get-address-info-by-cep.ts b/core/out/typescript/get-address-info-by-cep.ts new file mode 100644 index 000000000..aef1e1530 --- /dev/null +++ b/core/out/typescript/get-address-info-by-cep.ts @@ -0,0 +1,151 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: get-address-info-by-cep +// content: daa7ccc2e0b6 +import { jsonStringField } from "./lib/json.ts"; +import { + GetAddressInfoByCepNotFoundError, + GetAddressInfoByCepValidationError, +} from "./errors.ts"; +import type { + Capabilities, + HttpRequest, + HttpResponse, +} from "./capabilities.ts"; +import { DEFAULT_CAPABILITIES, raceFirstSome } from "./capabilities.ts"; + +export type AddressInfo = { + /** The 8 digit CEP, no mask. */ + readonly cep: string; + /** Two letter state code, e.g. "SP". */ + readonly state: string; + /** City name. */ + readonly city: string; + /** Neighborhood name, empty when the CEP covers a whole city. */ + readonly neighborhood: string; + /** Street name, empty when the CEP covers a whole city. */ + readonly street: string; +}; + +/** + * One GET, retried the way the published package retries: twice more, 250 ms apart. + */ +async function getWithRetry( + url: string, + env: Capabilities, +): Promise { + for (let attempt = 0; attempt < 3; attempt++) { + if (attempt > 0) { + await env.sleep(250); + } + const response: HttpResponse | undefined = await env.request({ + method: "GET", + url: url, + headers: [], + body: "", + timeoutMillis: 10000, + }); + if (response !== undefined) { + return response; + } + } + return undefined; +} + +/** + * Whether the status is a 2xx. + */ +function isOk(status: number): boolean { + return status >= 200 && status < 300; +} + +/** + * ViaCEP answers a JSON object, and marks an unknown CEP with `"erro"`. + */ +async function fetchViaCep( + cep: string, + env: Capabilities, +): Promise { + const response: HttpResponse | undefined = await getWithRetry( + "https://viacep.com.br/ws/" + cep + "/json/", + env, + ); + if (response === undefined || !isOk(response!.status)) { + return undefined; + } + const code: string = jsonStringField(response!.body, "cep") ?? ""; + if (code === "") { + return undefined; + } + return { + cep: code.replace(/[^0-9]/gu, ""), + state: jsonStringField(response!.body, "uf") ?? "", + city: jsonStringField(response!.body, "localidade") ?? "", + neighborhood: jsonStringField(response!.body, "bairro") ?? "", + street: jsonStringField(response!.body, "logradouro") ?? "", + }; +} + +/** + * BrasilAPI answers 404 for an unknown CEP. + */ +async function fetchBrasilApi( + cep: string, + env: Capabilities, +): Promise { + const response: HttpResponse | undefined = await getWithRetry( + "https://brasilapi.com.br/api/cep/v1/" + cep, + env, + ); + if (response === undefined || !isOk(response!.status)) { + return undefined; + } + const code: string = jsonStringField(response!.body, "cep") ?? ""; + if (code === "") { + return undefined; + } + return { + cep: code.replace(/[^0-9]/gu, ""), + state: jsonStringField(response!.body, "state") ?? "", + city: jsonStringField(response!.body, "city") ?? "", + neighborhood: jsonStringField(response!.body, "neighborhood") ?? "", + street: jsonStringField(response!.body, "street") ?? "", + }; +} + +/** + * The address of a CEP, from the first service that answers. + * + * The two services are queried concurrently and the first answer wins; the losing request may + * still finish, and its answer is dropped, which is why only idempotent GETs belong here. Each + * request is retried twice, 250 ms apart, exactly as the published package does. Turning a host + * value into the 8 digits this takes is the DX's job. + */ +export async function getAddressInfoByCep(cep: string): Promise { + return await getAddressInfoByCepWith(cep, DEFAULT_CAPABILITIES); +} + +/** + * `getAddressInfoByCep`, taking its capabilities explicitly. + * + * The public `getAddressInfoByCep` calls this with the platform's defaults. Pass your own to + * supply a clock, a source of randomness or an HTTP client — which is what the + * differential conformance driver does to make a run reproducible. + */ +export async function getAddressInfoByCepWith( + cep: string, + env: Capabilities, +): Promise { + if (!/^[0-9]{8}$/u.test(cep)) { + throw new GetAddressInfoByCepValidationError("CEP inv\u00e1lido"); + } + const address: AddressInfo | undefined = await raceFirstSome([ + async (): Promise => await fetchViaCep(cep, env), + async (): Promise => + await fetchBrasilApi(cep, env), + ]); + if (address === undefined) { + throw new GetAddressInfoByCepNotFoundError("CEP n\u00e3o encontrado"); + } + return address!; +} diff --git a/core/out/typescript/get-holidays.ts b/core/out/typescript/get-holidays.ts new file mode 100644 index 000000000..7c8b8dd4b --- /dev/null +++ b/core/out/typescript/get-holidays.ts @@ -0,0 +1,149 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: get-holidays +// content: b4d687034433 +import { civilDate10 } from "./lib/civil.ts"; +import { easterDayOfMarch } from "./lib/easter.ts"; +import { daysFromCivil } from "./std/date.ts"; + +export type Holiday = { + /** The holiday name in Brazilian Portuguese. */ + readonly name: string; + /** The day it falls on. */ + readonly date: number; + /** How it is observed. */ + readonly type: "national" | "optional" | "religious" | "state"; +}; + +/** + * The Brazilian national holidays of a year, sorted by date. + * + * The order is the one the published package produces: the fixed holidays in statutory order, + * then the Easter-derived ones, sorted by date with a stable sort, so two holidays on the same + * day keep the order they were built in. State holidays are not part of this pilot. + */ +export function getHolidays(year: number): readonly Holiday[] { + let holidays: Holiday[] = []; + holidays.push({ + name: "Ano novo", + date: + (year < 1 || year > 9999 || 1 < 1 || 1 > 31 + ? undefined + : daysFromCivil(year, 1, 1)) ?? Math.min(Math.max(0, -719162), 2932896), + type: "national", + }); + holidays.push({ + name: "Tiradentes", + date: + (year < 1 || year > 9999 || 21 < 1 || 21 > 30 + ? undefined + : daysFromCivil(year, 4, 21)) ?? + Math.min(Math.max(0, -719162), 2932896), + type: "national", + }); + holidays.push({ + name: "Dia do trabalhador", + date: + (year < 1 || year > 9999 || 1 < 1 || 1 > 31 + ? undefined + : daysFromCivil(year, 5, 1)) ?? Math.min(Math.max(0, -719162), 2932896), + type: "national", + }); + holidays.push({ + name: "Independ\u00eancia do Brasil", + date: + (year < 1 || year > 9999 || 7 < 1 || 7 > 30 + ? undefined + : daysFromCivil(year, 9, 7)) ?? Math.min(Math.max(0, -719162), 2932896), + type: "national", + }); + holidays.push({ + name: "Nossa Senhora Aparecida", + date: + (year < 1 || year > 9999 || 12 < 1 || 12 > 31 + ? undefined + : daysFromCivil(year, 10, 12)) ?? + Math.min(Math.max(0, -719162), 2932896), + type: "national", + }); + holidays.push({ + name: "Finados", + date: + (year < 1 || year > 9999 || 2 < 1 || 2 > 30 + ? undefined + : daysFromCivil(year, 11, 2)) ?? + Math.min(Math.max(0, -719162), 2932896), + type: "national", + }); + holidays.push({ + name: "Proclama\u00e7\u00e3o da Rep\u00fablica", + date: + (year < 1 || year > 9999 || 15 < 1 || 15 > 30 + ? undefined + : daysFromCivil(year, 11, 15)) ?? + Math.min(Math.max(0, -719162), 2932896), + type: "national", + }); + holidays.push({ + name: "Natal", + date: + (year < 1 || year > 9999 || 25 < 1 || 25 > 31 + ? undefined + : daysFromCivil(year, 12, 25)) ?? + Math.min(Math.max(0, -719162), 2932896), + type: "national", + }); + if (year >= 2024) { + holidays.push({ + name: "Dia da Consci\u00eancia Negra", + date: + (year < 1 || year > 9999 || 20 < 1 || 20 > 30 + ? undefined + : daysFromCivil(year, 11, 20)) ?? + Math.min(Math.max(0, -719162), 2932896), + type: "national", + }); + } + const _inl72DayOfMarch: number = easterDayOfMarch(year); + const _inl74Result: number = + _inl72DayOfMarch <= 31 + ? ((year < 1 || + year > 9999 || + _inl72DayOfMarch < 1 || + _inl72DayOfMarch > 31 + ? undefined + : daysFromCivil(year, 3, _inl72DayOfMarch)) ?? + Math.min(Math.max(0, -719162), 2932896)) + : civilDate10(year, _inl72DayOfMarch - 31); + const easter: number = _inl74Result; + holidays.push({ + name: "Carnaval (ter\u00e7a-feira)", + date: + (easter + -47 >= -719162 && easter + -47 <= 2932896 + ? easter + -47 + : undefined) ?? easter, + type: "optional", + }); + holidays.push({ + name: "Sexta-feira Santa", + date: + (easter + -2 >= -719162 && easter + -2 <= 2932896 + ? easter + -2 + : undefined) ?? easter, + type: "national", + }); + holidays.push({ name: "P\u00e1scoa", date: easter, type: "religious" }); + holidays.push({ + name: "Corpus Christi", + date: + (easter + 60 >= -719162 && easter + 60 <= 2932896 + ? easter + 60 + : undefined) ?? easter, + type: "optional", + }); + return [...holidays].sort((a, b) => { + const left = ((holiday: Holiday): number => holiday.date)(a); + const right = ((holiday: Holiday): number => holiday.date)(b); + return left < right ? -1 : left > right ? 1 : 0; + }); +} diff --git a/core/out/typescript/is-business-day.ts b/core/out/typescript/is-business-day.ts new file mode 100644 index 000000000..fd505c277 --- /dev/null +++ b/core/out/typescript/is-business-day.ts @@ -0,0 +1,39 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: is-business-day +// content: 3605fae387a8 +import { getHolidays } from "./get-holidays.ts"; +import { yearFromDays } from "./std/date.ts"; + +/** + * Whether a date is a Brazilian business day (dia útil). + * + * A day is not a business day when it falls on a weekend, or when it is one of the holidays + * `getHolidays` lists for its year. `includeOptional` decides whether the ponto facultativo + * entries (Carnaval, Corpus Christi) count; the published package defaults it to `true`, and + * supplying that default is the DX's job. + * + * Only the years 1900 to 2099 are supported, the range the holiday rules are stated for. + */ +export function isBusinessDay( + value: number, + includeOptional: boolean, +): boolean { + const year: number = yearFromDays(value); + if (year < 1900 || year > 2099) { + return false; + } + const weekday: number = ((((value + 3) % 7) + 7) % 7) + 1; + if (weekday === 6 || weekday === 7) { + return false; + } + for (const holiday of getHolidays(year)) { + if (!includeOptional && holiday.type === "optional") { + continue; + } + if ((holiday.date < value ? -1 : holiday.date > value ? 1 : 0) === 0) { + return false; + } + } + return true; +} diff --git a/core/out/typescript/is-valid-cnpj.ts b/core/out/typescript/is-valid-cnpj.ts new file mode 100644 index 000000000..4999ddbdf --- /dev/null +++ b/core/out/typescript/is-valid-cnpj.ts @@ -0,0 +1,37 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: is-valid-cnpj +// content: 611130f5f12f +import { hasLetter, hasValidCnpjChecksum, isRepeatedCnpj } from "./lib/cnpj.ts"; + +/** + * Validates a CNPJ (Cadastro Nacional da Pessoa Jurídica), numeric or alphanumeric. + * + * Version `"2"` accepts the alphanumeric format as well; a value with no letters is always read + * as the numeric one, which is also where the reserved repeated numbers are rejected. Mapping a + * missing or unexpected `options.version` onto `"1"` is the DX's job. + */ +export function isValidCnpj(cnpj: string, version: "1" | "2"): boolean { + const trimmed: string = cnpj.trim(); + if (version === "2") { + const cleaned: string = cnpj.replace(/[^0-9A-Za-z]/gu, "").toUpperCase(); + if (hasLetter(cleaned) && cleaned.length === 14) { + return ( + /^[0-9A-Z]{2}[\x09-\x0d \--/\u00a0\u1680\u2000-\u200a\u2028-\u2029\u202f\u205f\u3000\ufeff]*[0-9A-Z]{3}[\x09-\x0d \--/\u00a0\u1680\u2000-\u200a\u2028-\u2029\u202f\u205f\u3000\ufeff]*[0-9A-Z]{3}[\x09-\x0d \--/\u00a0\u1680\u2000-\u200a\u2028-\u2029\u202f\u205f\u3000\ufeff]*[0-9A-Z]{4}[\x09-\x0d \--/\u00a0\u1680\u2000-\u200a\u2028-\u2029\u202f\u205f\u3000\ufeff]*[0-9]{2}$/u.test( + trimmed.replace(/[a-z]/gu, (scalar) => scalar.toUpperCase()), + ) && hasValidCnpjChecksum(cleaned) + ); + } + } + const numeric: string = cnpj.replace(/[^0-9]/gu, ""); + if (numeric.length !== 14) { + return false; + } + return ( + /^[0-9]{2}[\x09-\x0d \--/\u00a0\u1680\u2000-\u200a\u2028-\u2029\u202f\u205f\u3000\ufeff]*[0-9]{3}[\x09-\x0d \--/\u00a0\u1680\u2000-\u200a\u2028-\u2029\u202f\u205f\u3000\ufeff]*[0-9]{3}[\x09-\x0d \--/\u00a0\u1680\u2000-\u200a\u2028-\u2029\u202f\u205f\u3000\ufeff]*[0-9]{4}[\x09-\x0d \--/\u00a0\u1680\u2000-\u200a\u2028-\u2029\u202f\u205f\u3000\ufeff]*[0-9]{2}$/u.test( + trimmed, + ) && + !isRepeatedCnpj(numeric) && + hasValidCnpjChecksum(numeric) + ); +} diff --git a/core/out/typescript/is-valid-cpf.ts b/core/out/typescript/is-valid-cpf.ts new file mode 100644 index 000000000..fbabd0b88 --- /dev/null +++ b/core/out/typescript/is-valid-cpf.ts @@ -0,0 +1,32 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: is-valid-cpf +// content: dae364e11fb3 +import { cpfCheckDigit, cpfCheckDigit1, isRepeated } from "./lib/cpf.ts"; + +/** + * Validates a CPF (Cadastro de Pessoas Físicas). + * + * The core takes the value as written, accepting the usual mask characters; turning a host value + * into a string is the DX's job. + */ +export function isValidCpf(cpf: string): boolean { + if ( + !/^[0-9]{3}[\x09-\x0d \--/\u00a0\u1680\u2000-\u200a\u2028-\u2029\u202f\u205f\u3000\ufeff]*[0-9]{3}[\x09-\x0d \--/\u00a0\u1680\u2000-\u200a\u2028-\u2029\u202f\u205f\u3000\ufeff]*[0-9]{3}[\x09-\x0d \--/\u00a0\u1680\u2000-\u200a\u2028-\u2029\u202f\u205f\u3000\ufeff]*[0-9]{2}$/u.test( + cpf.trim(), + ) + ) { + return false; + } + const digits: string = cpf.replace(/[^0-9]/gu, ""); + if (digits.length !== 11) { + return false; + } + if (isRepeated(digits)) { + return false; + } + return ( + digits.charCodeAt(9) - 48 === cpfCheckDigit(digits) && + digits.charCodeAt(10) - 48 === cpfCheckDigit1(digits) + ); +} diff --git a/core/out/typescript/lib/civil.ts b/core/out/typescript/lib/civil.ts new file mode 100644 index 000000000..e83d27908 --- /dev/null +++ b/core/out/typescript/lib/civil.ts @@ -0,0 +1,16 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/civil +// content: c2aca1eb50dc +import { daysFromCivil } from "../std/date.ts"; + +/** + * A fixed day of a year, with the unreachable fallback named once. + */ +export function civilDate10(year: number, day: number): number { + return ( + (year < 1 || year > 9999 || day < 1 || day > 30 + ? undefined + : daysFromCivil(year, 4, day)) ?? Math.min(Math.max(0, -719162), 2932896) + ); +} diff --git a/core/out/typescript/lib/cnpj.ts b/core/out/typescript/lib/cnpj.ts new file mode 100644 index 000000000..a2ccc7806 --- /dev/null +++ b/core/out/typescript/lib/cnpj.ts @@ -0,0 +1,91 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/cnpj +// content: 78a3fa4783b9 +import { randomBelow } from "./random.ts"; +import type { Capabilities } from "../capabilities.ts"; +import { raceFirstSome } from "../capabilities.ts"; + +const libCnpjTable1: readonly number[] = [5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]; + +const libCnpjTable2: readonly number[] = [ + 6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2, +]; + +/** + * A random numeric CNPJ base: an 8-digit root and a 4-digit branch, each digit drawn + * independently — matches the published `generateCnpj()` called with no branch, where an unset + * branch also draws those 4 digits at random. Twelve separate draws, not a loop, is what lets the + * result stay exactly 12 digits long. + */ +export function randomCnpjBase(env: Capabilities): string { + return ( + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + ); +} + +/** + * The check digit of a CNPJ base, under the rule both versions share. + */ +export function cnpjCheckDigit( + cnpj: string, + weights: readonly number[], +): number { + let sum: number = 0; + for (let index = 0; index < weights.length; index++) { + sum = sum + (cnpj.charCodeAt(index) - 48) * weights[index]; + } + const remainder: number = sum % 11; + return remainder < 2 ? 0 : 11 - remainder; +} + +/** + * Whether the value holds at least one upper cased ASCII letter. + * + * The scan reads positions rather than materializing the scalars, which the checked accessor + * makes safe without a proof about the length. + */ +export function hasLetter(value: string): boolean { + for (let index = 0; index < value.length; index++) { + const point: number = + (index < value.length ? value.charCodeAt(index) : undefined) ?? 0; + if (point >= 65 && point <= 90) { + return true; + } + } + return false; +} + +/** + * Whether both check digits of a 14 character CNPJ match its base. + */ +export function hasValidCnpjChecksum(cnpj: string): boolean { + return ( + cnpj.charCodeAt(12) - 48 === cnpjCheckDigit(cnpj, libCnpjTable1) && + cnpj.charCodeAt(13) - 48 === cnpjCheckDigit(cnpj, libCnpjTable2) + ); +} + +/** + * Whether every character of a 14 character value is the same one. + */ +export function isRepeatedCnpj(value: string): boolean { + const first: number = value.charCodeAt(0); + for (let index = 1; index < 14; index++) { + if (value.charCodeAt(index) !== first) { + return false; + } + } + return true; +} diff --git a/core/out/typescript/lib/cpf.ts b/core/out/typescript/lib/cpf.ts new file mode 100644 index 000000000..d9ba24603 --- /dev/null +++ b/core/out/typescript/lib/cpf.ts @@ -0,0 +1,63 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/cpf +// content: 5b319831cfe4 +import { randomBelow } from "./random.ts"; +import type { Capabilities } from "../capabilities.ts"; +import { raceFirstSome } from "../capabilities.ts"; + +/** + * A random CPF base: 8 digits plus a região fiscal digit, each drawn independently — matches the + * published `generateCpf()` called with no state, where an unset state also draws that 9th digit + * at random. Nine separate draws, not a loop, is what lets the result stay exactly 9 digits long. + */ +export function randomCpfBase(env: Capabilities): string { + return ( + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + + randomBelow(env).toString() + ); +} + +/** + * The check digit of a CPF base, under the Receita Federal rule (weights 10..2 and 11..2). + */ +export function cpfCheckDigit(cpf: string): number { + let sum: number = 0; + for (let index = 0; index < 9; index++) { + sum = sum + (cpf.charCodeAt(index) - 48) * (10 - index); + } + const remainder: number = sum % 11; + return remainder < 2 ? 0 : 11 - remainder; +} + +/** + * The check digit of a CPF base, under the Receita Federal rule (weights 10..2 and 11..2). + */ +export function cpfCheckDigit1(cpf: string): number { + let sum: number = 0; + for (let index = 0; index < 10; index++) { + sum = sum + (cpf.charCodeAt(index) - 48) * (11 - index); + } + const remainder: number = sum % 11; + return remainder < 2 ? 0 : 11 - remainder; +} + +/** + * Whether every scalar of the value is the same one, e.g. "00000000000". + */ +export function isRepeated(value: string): boolean { + const first: number = value.charCodeAt(0); + for (let index = 1; index < 11; index++) { + if (value.charCodeAt(index) !== first) { + return false; + } + } + return true; +} diff --git a/core/out/typescript/lib/digits.ts b/core/out/typescript/lib/digits.ts new file mode 100644 index 000000000..cdcebe6f6 --- /dev/null +++ b/core/out/typescript/lib/digits.ts @@ -0,0 +1,18 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/digits +// content: 780bb464b89e +/** + * Whether every scalar of the value is the same one, for whatever length the caller proved — + * `isRepeated` and `isRepeatedCnpj` do the same check for one specific length; this one serves a + * generator that has to run it on a base shorter than the document it is building. + */ +export function isRepeatedRun(value: string): boolean { + const first: number = value.charCodeAt(0); + for (let index = 1; index < value.length; index++) { + if (value.charCodeAt(index) !== first) { + return false; + } + } + return true; +} diff --git a/core/out/typescript/lib/easter.ts b/core/out/typescript/lib/easter.ts new file mode 100644 index 000000000..8456d19a8 --- /dev/null +++ b/core/out/typescript/lib/easter.ts @@ -0,0 +1,26 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/easter +// content: c99e5cf5c72b +import { civilDate10 } from "./civil.ts"; + +/** + * The day of March (1 to 31) or April (32 to 56) Easter falls on, as a day-of-March offset. + */ +export function easterDayOfMarch(year: number): number { + const a: number = year % 19; + const b: number = Math.trunc(year / 100); + const c: number = year % 100; + const d: number = Math.trunc(b / 4); + const e: number = b % 4; + const h: number = (19 * a + b - d - 6 + 15) % 30; + const i: number = Math.trunc(c / 4); + const k: number = c % 4; + const l: number = (32 + 2 * e + 2 * i - h - k) % 7; + const m: number = Math.trunc((a + 11 * h + 22 * l) / 451); + const day: number = h + l - 7 * m + 114; + return Math.min( + Math.max((day % 31) + 1 + (Math.trunc(day / 31) - 3) * 31, 22), + 56, + ); +} diff --git a/core/out/typescript/lib/format.ts b/core/out/typescript/lib/format.ts new file mode 100644 index 000000000..1bd9f3eab --- /dev/null +++ b/core/out/typescript/lib/format.ts @@ -0,0 +1,47 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/format +// content: 0038a81c2e4d +/** + * How many scalars of the value a pattern consumes. + */ +export function patternSlots(pattern: string): number { + let slots: number = 0; + for (let index = 0; index < pattern.length; index++) { + const symbol: string = pattern[index] ?? ""; + if (symbol === "0" || symbol === "*") { + slots = slots + 1; + } + } + return slots; +} + +/** + * Formats a value against a pattern, optionally left padding it with zeros first. + */ +export function formatWithPattern( + value: string, + pattern: string, + pad: boolean, +): string { + const padded: string = pad + ? value.padStart(patternSlots(pattern), "0") + : value; + let out: string = ""; + let taken: number = 0; + for (let index = 0; index < pattern.length; index++) { + const symbol: string = pattern[index] ?? ""; + if (symbol === "0" || symbol === "*") { + if (taken >= padded.length) { + return out; + } + out = out + (symbol === "*" ? "*" : (padded[taken] ?? "")); + taken = taken + 1; + } else { + if (taken < padded.length) { + out = out + symbol; + } + } + } + return out; +} diff --git a/core/out/typescript/lib/json.ts b/core/out/typescript/lib/json.ts new file mode 100644 index 000000000..8b63c0264 --- /dev/null +++ b/core/out/typescript/lib/json.ts @@ -0,0 +1,131 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/json +// content: 02cc75dfd626 +/** + * Whether `needle` occurs in `points` at `start`. + */ +function matchesAt( + points: readonly number[], + needle: readonly number[], + start: number, +): boolean { + for (let offset = 0; offset < needle.length; offset++) { + if ((points[start + offset] ?? -1) !== (needle[offset] ?? -2)) { + return false; + } + } + return true; +} + +/** + * Whether a code point is JSON whitespace. + */ +function isSpace(point: number): boolean { + return point === 32 || point === 9 || point === 10 || point === 13; +} + +/** + * The hexadecimal value of four scalars, for a `\uXXXX` escape. + */ +function hexValue(points: readonly number[], start: number): number { + let value: number = 0; + for (let offset = 0; offset < 4; offset++) { + const point: number = points[start + offset] ?? 48; + let digit: number = 0; + if (point >= 48 && point <= 57) { + digit = point - 48; + } else { + if (point >= 97 && point <= 102) { + digit = point - 87; + } else { + if (point >= 65 && point <= 70) { + digit = point - 55; + } + } + } + value = value * 16 + digit; + } + return Math.min(value, 65535); +} + +/** + * The string value of a top-level JSON field, or absent when the field is missing or is not a + * string. Escapes are decoded; a surrogate pair is left as its two escaped halves, which no CEP + * provider emits. + */ +export function jsonStringField(body: string, key: string): string | undefined { + const points: readonly number[] = Array.from( + body, + (scalar) => scalar.codePointAt(0)!, + ); + const needle: readonly number[] = Array.from( + '"' + key + '"', + (scalar) => scalar.codePointAt(0)!, + ); + for (let index = 0; index < points.length; index++) { + if (!matchesAt(points, needle, index)) { + continue; + } + let cursor: number = index + needle.length; + for (let skip = 0; skip < 8; skip++) { + if (isSpace(points[cursor] ?? 0)) { + cursor = cursor + 1; + } + } + if ((points[cursor] ?? 0) !== 58) { + continue; + } + cursor = cursor + 1; + for (let skip = 0; skip < 8; skip++) { + if (isSpace(points[cursor] ?? 0)) { + cursor = cursor + 1; + } + } + if ((points[cursor] ?? 0) !== 34) { + continue; + } + cursor = cursor + 1; + let out: number[] = []; + for (let step = 0; step < points.length; step++) { + const point: number = points[cursor] ?? -1; + if (point === -1 || point === 34) { + return out.map((point) => String.fromCodePoint(point)).join(""); + } + if (point === 92) { + const escaped: number = points[cursor + 1] ?? -1; + if (escaped === 110) { + out.push(10); + cursor = cursor + 2; + } else { + if (escaped === 116) { + out.push(9); + cursor = cursor + 2; + } else { + if (escaped === 114) { + out.push(13); + cursor = cursor + 2; + } else { + if (escaped === 117) { + out.push(hexValue(points, cursor + 2)); + cursor = cursor + 6; + } else { + if (escaped >= 0) { + out.push(escaped); + cursor = cursor + 2; + } else { + cursor = cursor + 1; + } + } + } + } + } + } else { + out.push(point); + cursor = cursor + 1; + } + } + return out.map((point) => String.fromCodePoint(point)).join(""); + } + return undefined; +} diff --git a/core/out/typescript/lib/random.ts b/core/out/typescript/lib/random.ts new file mode 100644 index 000000000..75d58289b --- /dev/null +++ b/core/out/typescript/lib/random.ts @@ -0,0 +1,23 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: lib/random +// content: 2101f5409061 +import type { Capabilities } from "../capabilities.ts"; +import { raceFirstSome } from "../capabilities.ts"; + +/** + * A uniform integer in `[0, bound)`, by rejection sampling rather than `% bound`: the modulo of a + * fixed-width draw is biased whenever `bound` does not divide 2^32 evenly, and that bias would + * have to match, digit for digit, across three unrelated standard libraries to stay invisible. + * Rejecting the biased tail of the draw removes it instead. + */ +export function randomBelow(env: Capabilities): number { + const limit: number = 4294967290; + for (let attempt = 0; attempt < 32; attempt++) { + const draw: number = env.nextU32(); + if (draw < limit) { + return draw % 10; + } + } + return env.nextU32() % 10; +} diff --git a/core/out/typescript/std/date.ts b/core/out/typescript/std/date.ts new file mode 100644 index 000000000..a7094d4a5 --- /dev/null +++ b/core/out/typescript/std/date.ts @@ -0,0 +1,69 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: std/date +// content: 8041c981a090 +/** + * Floor division, which the calendar algorithms need for negative years. + */ +function floorDiv1(value: number): number { + const quotient: number = Math.trunc(value / 400); + if (value < 0 && quotient * 400 !== value) { + return quotient - 1; + } + return quotient; +} + +/** + * Days since 1970-01-01 for a year, month and day already known to be a real date. + */ +export function daysFromCivil( + year: number, + month: number, + day: number, +): number { + const shifted: number = month <= 2 ? year - 1 : year; + const era: number = floorDiv1(shifted); + const yearOfEra: number = shifted - era * 400; + const monthTerm: number = month > 2 ? month - 3 : month + 9; + const dayOfYear: number = Math.trunc((153 * monthTerm + 2) / 5) + day - 1; + const dayOfEra: number = + yearOfEra * 365 + + Math.trunc(yearOfEra / 4) - + Math.trunc(yearOfEra / 100) + + dayOfYear; + return Math.min(Math.max(era * 146097 + dayOfEra - 719468, -719162), 2932896); +} + +/** + * Floor division, which the calendar algorithms need for negative years. + */ +export function floorDiv(value: number): number { + const quotient: number = Math.trunc(value / 146097); + if (value < 0 && quotient * 146097 !== value) { + return quotient - 1; + } + return quotient; +} + +/** + * The year of a date given as days since 1970-01-01. + */ +export function yearFromDays(days: number): number { + const shifted: number = days + 719468; + const era: number = floorDiv(shifted); + const dayOfEra: number = shifted - era * 146097; + const yearOfEra: number = Math.trunc( + (dayOfEra - + Math.trunc(dayOfEra / 1460) + + Math.trunc(dayOfEra / 36524) - + Math.trunc(dayOfEra / 146096)) / + 365, + ); + const year: number = yearOfEra + era * 400; + const dayOfYear: number = + dayOfEra - + (365 * yearOfEra + Math.trunc(yearOfEra / 4) - Math.trunc(yearOfEra / 100)); + const monthPrime: number = Math.trunc((5 * dayOfYear + 2) / 153); + const month: number = monthPrime < 10 ? monthPrime + 3 : monthPrime - 9; + return Math.min(Math.max(month <= 2 ? year + 1 : year, 1), 9999); +} diff --git a/core/package.json b/core/package.json new file mode 100644 index 000000000..f2a8d2570 --- /dev/null +++ b/core/package.json @@ -0,0 +1,18 @@ +{ + "name": "@brazilian-utils/core", + "version": "0.1.0", + "private": true, + "description": "The single-source core of Brazilian Utils: each utility is written once in the engine's restricted subset and generated for TypeScript, Python and Go.", + "license": "MIT", + "type": "module", + "engines": { + "node": ">=22.12.0" + }, + "scripts": { + "check": "node ../engine/src/cli.ts check --project .", + "build": "node ../engine/src/cli.ts build --project . && node ../engine/src/cli.ts build --project . --no-idioms", + "dump": "node ../engine/src/cli.ts dump --project .", + "conformance": "node --import ./conformance/sloppy-imports.mjs ./conformance/run.ts", + "verify": "node ../engine/scripts/verify.ts ." + } +} diff --git a/core/scripts/survey.ts b/core/scripts/survey.ts new file mode 100644 index 000000000..7c778a26d --- /dev/null +++ b/core/scripts/survey.ts @@ -0,0 +1,130 @@ +#!/usr/bin/env node +/** + * Classifies every utility of the published package by the feature it needs from the engine, so + * "how many utilities can be authored once" has a measured answer rather than an opinion. + * + * The classification is syntactic and deliberately conservative: it reads the utility and + * everything it imports from `src/_internals`, and reports what it finds. + */ + +import { existsSync, readFileSync, readdirSync, statSync } from "node:fs"; +import { join, resolve } from "node:path"; + +const PACKAGE_SRC = resolve(import.meta.dirname, "..", "..", "src"); + +type Feature = + | "strings" + | "ascii" + | "regex" + | "ranges" + | "dataset" + | "collation" + | "dates" + | "decimal" + | "float" + | "random" + | "http" + | "unicode" + | "recursion" + | "maps" + | "unions"; + +const DETECTORS: { feature: Feature; test: RegExp }[] = [ + { feature: "regex", test: /\/\^|RegExp|\.test\(|replaceAll\(\//u }, + { feature: "ascii", test: /charCodeAt|charAt|padStart|toUpperCase|toLowerCase/u }, + { feature: "dates", test: /new Date|getFullYear|getMonth|getDay\(/u }, + { feature: "float", test: /Number\(|parseFloat|Math\.(round|floor|ceil|abs|pow)/u }, + { feature: "decimal", test: /toFixed|Intl\.NumberFormat|precision/u }, + { feature: "random", test: /Math\.random|randomInt|pickRandom/u }, + { feature: "http", test: /fetch\(|fetchWithRetry/u }, + { feature: "unicode", test: /normalize\(|localeCompare|\\p\{/u }, + { feature: "maps", test: /new Map\(|new Set\(|Object\.entries|Object\.keys/u }, + { feature: "dataset", test: /from "\.\.\/_internals\/constants|constants"|CITIES|BANKS|CNAE/u }, + { feature: "collation", test: /\.sort\(|localeCompare/u }, + { feature: "unions", test: /kind:|type:\s*"/u }, +]; + +const utilities = readdirSync(PACKAGE_SRC) + .filter((entry) => statSync(join(PACKAGE_SRC, entry)).isDirectory() && entry !== "_internals") + .sort(); + +const rows: { name: string; features: Feature[]; lines: number }[] = []; + +for (const utility of utilities) { + const file = join(PACKAGE_SRC, utility, `${utility}.ts`); + if (!existsSync(file)) continue; + const source = readFileSync(file, "utf8"); + const constants = join(PACKAGE_SRC, utility, "constants.ts"); + const combined = source + (existsSync(constants) ? readFileSync(constants, "utf8") : ""); + const features = DETECTORS.filter((detector) => detector.test.test(combined)).map((detector) => detector.feature); + if (existsSync(constants)) features.push("dataset"); + rows.push({ + name: utility, + features: [...new Set(["strings" as Feature, ...features])].sort(), + lines: source.split("\n").filter((line) => line.trim() !== "" && !line.trim().startsWith("*")).length, + }); +} + +const counts = new Map(); +for (const row of rows) { + for (const feature of row.features) counts.set(feature, (counts.get(feature) ?? 0) + 1); +} + +/** What the engine supports today, and what a utility needing it is blocked by. */ +const SUPPORT: Record = { + strings: "supported", + ascii: "supported", + regex: "supported (explicit classes only)", + ranges: "supported", + dataset: "supported as constant tables; a baked dataset type is not admitted yet", + collation: "supported (scalar order); locale collation is not admitted", + dates: "supported (CivilDate)", + decimal: "supported (fixed scale)", + float: "supported", + random: "supported (PCG32 in source over random.nextU32)", + http: "supported", + unicode: "blocked: normalization and locale case mapping are out of scope", + recursion: "blocked: not admitted in this phase", + maps: "blocked: Map and Set are not admitted yet", + unions: "blocked: discriminated unions are not admitted yet", +}; + +const blocked = rows.filter((row) => row.features.some((feature) => SUPPORT[feature].startsWith("blocked"))); + +const lines = [ + "# Survey of the published package", + "", + "Generated by `node scripts/survey.ts`. The classification is syntactic and conservative: it", + "reads each utility and reports the features it appears to need. It answers the question the", + "engine exists to answer — how many utilities could be authored once — with a number rather", + "than a feeling.", + "", + `**${rows.length} utilities.** ${rows.length - blocked.length} use only features the engine supports today;`, + `${blocked.length} touch something that is not admitted yet.`, + "", + "## Features by utility count", + "", + "| feature | utilities | status |", + "| --- | --- | --- |", + ...[...counts.entries()] + .sort(([, left], [, right]) => right - left) + .map(([feature, count]) => `| ${feature} | ${count} | ${SUPPORT[feature]} |`), + "", + "## Blocked utilities", + "", + "| utility | blocked by |", + "| --- | --- |", + ...blocked.map( + (row) => + `| \`${row.name}\` | ${row.features.filter((feature) => SUPPORT[feature].startsWith("blocked")).join(", ")} |`, + ), + "", + "## Every utility", + "", + "| utility | lines | features |", + "| --- | --- | --- |", + ...rows.map((row) => `| \`${row.name}\` | ${row.lines} | ${row.features.join(", ")} |`), + "", +]; + +process.stdout.write(`${lines.join("\n")}\n`); diff --git a/core/source/format-cnpj.ts b/core/source/format-cnpj.ts new file mode 100644 index 000000000..43f425c55 --- /dev/null +++ b/core/source/format-cnpj.ts @@ -0,0 +1,33 @@ +import { formatWithPattern } from "./lib/format"; +import { keepAlphanumeric, keepDigits } from "./lib/digits"; +import { type CnpjVersion } from "./is-valid-cnpj"; + +const PATTERN = "00.000.000/0000-00"; + +/** + * The gov.br / Receita Federal display convention: the first two characters and the two check + * digits are hidden. + */ +const OBFUSCATED_PATTERN = "**.000.000/0000-**"; + +/** Options of `formatCnpj`, already normalized by the DX. */ +export type FormatCnpjOptions = { + /** Whether to left pad the value with zeros up to the number of slots in the pattern. */ + pad: boolean; + /** Which CNPJ format to read. */ + version: CnpjVersion; + /** Whether to hide the first two characters and the two check digits with `*`. */ + obfuscate: boolean; +}; + +/** + * Formats a CNPJ value as `00.000.000/0000-00`. + * + * The core takes a string and a fully normalized options record; reading a number, a missing + * options object or a truthy non-boolean is the DX's job. + */ +export function formatCnpj(value: string, options: FormatCnpjOptions): string { + const sanitized = options.version === "2" ? keepAlphanumeric(value) : keepDigits(value); + + return formatWithPattern(sanitized, options.obfuscate ? OBFUSCATED_PATTERN : PATTERN, options.pad); +} diff --git a/core/source/format-currency.ts b/core/source/format-currency.ts new file mode 100644 index 000000000..04db76d40 --- /dev/null +++ b/core/source/format-currency.ts @@ -0,0 +1,34 @@ +import { keepDigits } from "./lib/digits"; +import { groupThousands } from "./lib/format"; + +/** + * The separator after `R$`. + * + * CLDR pt-BR uses a non-breaking space here, and `Intl.NumberFormat` emits one; the published + * package replaces it with an ordinary space before returning, so that is what the core produces + * (docs/contracts.md records the measurement). + */ +const SPACE = 32; + +/** + * Formats an exact amount in Brazilian Real, with two decimal places. + * + * The separators are the ones Lei nº 9.069/1995 art. 1º prescribes and the CLDR pt-BR data uses: + * `.` between thousands, `,` before the centavos, and a non-breaking space after `R$`. A negative + * amount puts the sign before the symbol, `-R$ 10,50`, the shape `Intl.NumberFormat` produces. + * + * Turning a host value into an exact amount is the DX's job, and so is the rounding that + * conversion needs; see docs/contracts.md, which records exactly how the published package rounds. + */ +export function formatCurrency(value: Decimal<2>, symbol: boolean): string { + const negative = dec.isNegative(value); + const unscaled = dec.unscaled(dec.abs(value)); + const digits = String(unscaled).padStart(3, "0"); + const cut = Math.max(digits.length - 2, 0); + const whole = digits.slice(0, cut); + const cents = digits.slice(cut, digits.length); + const body = `${groupThousands(keepDigits(whole))},${cents}`; + const prefix = symbol ? `R$${str.fromCodePoints([SPACE])}` : ""; + + return negative ? `-${prefix}${body}` : `${prefix}${body}`; +} diff --git a/core/source/generate-cnpj.ts b/core/source/generate-cnpj.ts new file mode 100644 index 000000000..e2b4523bd --- /dev/null +++ b/core/source/generate-cnpj.ts @@ -0,0 +1,37 @@ +import { FIRST_WEIGHTS, SECOND_WEIGHTS, cnpjCheckDigit, randomCnpjBase } from "./lib/cnpj"; +import { isRepeatedRun } from "./lib/digits"; + +/** + * A 12-digit base repeats with probability 1 in 10^11; this many redraws leave a repeated base in + * the result with probability under 10^-88, the documented fallback a bounded `for` needs in + * place of the published implementation's unbounded `while`. + */ +const MAX_BASE_ATTEMPTS = 8; + +/** + * Generates a valid random CNPJ (Cadastro Nacional da Pessoa Jurídica) in the numeric format: 14 + * digits, under the check digit rule both CNPJ versions share. + * + * Matches the published `generateCnpj()` called with no options: a random 8-digit root and + * 4-digit branch (the "número de ordem"), redrawn while every digit of the 12-digit base is the + * same, followed by its two check digits. The alphanumeric version and a chosen branch are DX + * concerns layered on the same base and check digit rule, not a different generator. + */ +export function generateCnpj(): string { + let base = randomCnpjBase(); + + for (let attempt = 0; attempt < MAX_BASE_ATTEMPTS; attempt++) { + if (!isRepeatedRun(base)) { + break; + } + + base = randomCnpjBase(); + } + + // Both calls need a full 14-character value: the trailing positions the shorter weight list + // never reads are filled with a placeholder digit purely to satisfy that length. + const firstDigit = String(cnpjCheckDigit(`${base}00`, FIRST_WEIGHTS)); + const secondDigit = String(cnpjCheckDigit(`${base}${firstDigit}0`, SECOND_WEIGHTS)); + + return `${base}${firstDigit}${secondDigit}`; +} diff --git a/core/source/generate-cpf.ts b/core/source/generate-cpf.ts new file mode 100644 index 000000000..cbdecc712 --- /dev/null +++ b/core/source/generate-cpf.ts @@ -0,0 +1,38 @@ +import { cpfCheckDigit, randomCpfBase } from "./lib/cpf"; +import { isRepeatedRun } from "./lib/digits"; + +/** + * A 9-digit base repeats with probability 1 in 10^8; this many redraws leave a repeated base in + * the result with probability under 10^-64, the documented fallback a bounded `for` needs in + * place of the published implementation's unbounded `while`. + */ +const MAX_BASE_ATTEMPTS = 8; + +/** + * Generates a valid random CPF (Cadastro de Pessoas Físicas): 11 digits, under the check digit + * rule (weights 10..2 and 11..2) the Receita Federal's Manual de Preenchimento da e-Financeira, + * Anexo II specifies. + * + * Matches the published `generateCpf()` called with no state: a random 9-digit base — 8 digits + * plus a região fiscal digit, also drawn at random here — redrawn while every digit of it is the + * same, followed by its two check digits. The state code option is a DX concern: it only ever + * picks which digit the 9th position draws from, never how the rest of the document is built. + */ +export function generateCpf(): string { + let base = randomCpfBase(); + + for (let attempt = 0; attempt < MAX_BASE_ATTEMPTS; attempt++) { + if (!isRepeatedRun(base)) { + break; + } + + base = randomCpfBase(); + } + + // Both calls need a full 11-digit value: the trailing positions a `size` of 9 or 10 never + // reads are filled with a placeholder digit purely to satisfy that length. + const firstDigit = String(cpfCheckDigit(`${base}00`, 9)); + const secondDigit = String(cpfCheckDigit(`${base}${firstDigit}0`, 10)); + + return `${base}${firstDigit}${secondDigit}`; +} diff --git a/core/source/get-address-info-by-cep.ts b/core/source/get-address-info-by-cep.ts new file mode 100644 index 000000000..e8df91d16 --- /dev/null +++ b/core/source/get-address-info-by-cep.ts @@ -0,0 +1,133 @@ +import { jsonStringField } from "./lib/json"; +import { keepDigits } from "./lib/digits"; + +/** Base of every error this utility raises. */ +export class GetAddressInfoByCepError extends DomainError {} + +/** The value given is not a CEP. */ +export class GetAddressInfoByCepValidationError extends GetAddressInfoByCepError {} + +/** No CEP service knows this CEP, or none answered. */ +export class GetAddressInfoByCepNotFoundError extends GetAddressInfoByCepError {} + +/** The address of a CEP. */ +export type AddressInfo = { + /** The 8 digit CEP, no mask. */ + cep: string; + /** Two letter state code, e.g. "SP". */ + state: string; + /** City name. */ + city: string; + /** Neighborhood name, empty when the CEP covers a whole city. */ + neighborhood: string; + /** Street name, empty when the CEP covers a whole city. */ + street: string; +}; + +const CEP_FORMAT = /^[0-9]{8}$/; + +const ATTEMPTS = 3; + +const RETRY_DELAY_MILLIS = 250; + +const TIMEOUT_MILLIS = 10000; + +const OK = 200; + +const MULTIPLE_CHOICES = 300; + +/** One GET, retried the way the published package retries: twice more, 250 ms apart. */ +function getWithRetry(url: string): HttpResponse | undefined { + for (let attempt = 0; attempt < ATTEMPTS; attempt++) { + if (attempt > 0) { + clock.sleep(clock.millis(RETRY_DELAY_MILLIS)); + } + + const response = http.request({ + method: "GET", + url, + headers: [], + body: "", + timeoutMillis: TIMEOUT_MILLIS, + }); + + if (response !== undefined) { + return response; + } + } + + return undefined; +} + +/** Whether the status is a 2xx. */ +function isOk(status: number): boolean { + return status >= OK && status < MULTIPLE_CHOICES; +} + +/** ViaCEP answers a JSON object, and marks an unknown CEP with `"erro"`. */ +function fetchViaCep(cep: Digits): AddressInfo | undefined { + const response = getWithRetry(`https://viacep.com.br/ws/${cep}/json/`); + + if (response === undefined || !isOk(response.status)) { + return undefined; + } + + const code = jsonStringField(response.body, "cep") ?? ""; + + if (code === "") { + return undefined; + } + + return { + cep: keepDigits(code), + state: jsonStringField(response.body, "uf") ?? "", + city: jsonStringField(response.body, "localidade") ?? "", + neighborhood: jsonStringField(response.body, "bairro") ?? "", + street: jsonStringField(response.body, "logradouro") ?? "", + }; +} + +/** BrasilAPI answers 404 for an unknown CEP. */ +function fetchBrasilApi(cep: Digits): AddressInfo | undefined { + const response = getWithRetry(`https://brasilapi.com.br/api/cep/v1/${cep}`); + + if (response === undefined || !isOk(response.status)) { + return undefined; + } + + const code = jsonStringField(response.body, "cep") ?? ""; + + if (code === "") { + return undefined; + } + + return { + cep: keepDigits(code), + state: jsonStringField(response.body, "state") ?? "", + city: jsonStringField(response.body, "city") ?? "", + neighborhood: jsonStringField(response.body, "neighborhood") ?? "", + street: jsonStringField(response.body, "street") ?? "", + }; +} + +/** + * The address of a CEP, from the first service that answers. + * + * The two services are queried concurrently and the first answer wins; the losing request may + * still finish, and its answer is dropped, which is why only idempotent GETs belong here. Each + * request is retried twice, 250 ms apart, exactly as the published package does. Turning a host + * value into the 8 digits this takes is the DX's job. + */ +export function getAddressInfoByCep(cep: string): AddressInfo { + if (!CEP_FORMAT.test(cep)) { + throw new GetAddressInfoByCepValidationError("CEP inválido"); + } + + const address = task.race([(): AddressInfo | undefined => fetchViaCep(cep), (): AddressInfo | undefined => fetchBrasilApi(cep)]); + + if (address === undefined) { + throw new GetAddressInfoByCepNotFoundError("CEP não encontrado"); + } + + return address; +} diff --git a/core/source/get-holidays.ts b/core/source/get-holidays.ts new file mode 100644 index 000000000..faf66bbf3 --- /dev/null +++ b/core/source/get-holidays.ts @@ -0,0 +1,63 @@ +import { civilDate } from "./lib/civil"; +import { easterSunday } from "./lib/easter"; + +/** How a holiday is observed. */ +export type HolidayType = "national" | "optional" | "religious" | "state"; + +/** One Brazilian holiday. */ +export type Holiday = { + /** The holiday name in Brazilian Portuguese. */ + name: string; + /** The day it falls on. */ + date: CivilDate; + /** How it is observed. */ + type: HolidayType; +}; + +/** The first year Dia da Consciência Negra is a national holiday (Lei 14.759/2023). */ +const CONSCIENCIA_NEGRA_SINCE = 2024; + +/** + * The Brazilian national holidays of a year, sorted by date. + * + * The order is the one the published package produces: the fixed holidays in statutory order, + * then the Easter-derived ones, sorted by date with a stable sort, so two holidays on the same + * day keep the order they were built in. State holidays are not part of this pilot. + */ +export function getHolidays(year: IntRange<1900, 2099>): List { + let holidays: Holiday[] = []; + + holidays.push({ name: "Ano novo", date: civilDate(year, 1, 1), type: "national" }); + holidays.push({ name: "Tiradentes", date: civilDate(year, 4, 21), type: "national" }); + holidays.push({ name: "Dia do trabalhador", date: civilDate(year, 5, 1), type: "national" }); + holidays.push({ name: "Independência do Brasil", date: civilDate(year, 9, 7), type: "national" }); + holidays.push({ name: "Nossa Senhora Aparecida", date: civilDate(year, 10, 12), type: "national" }); + holidays.push({ name: "Finados", date: civilDate(year, 11, 2), type: "national" }); + holidays.push({ name: "Proclamação da República", date: civilDate(year, 11, 15), type: "national" }); + holidays.push({ name: "Natal", date: civilDate(year, 12, 25), type: "national" }); + + if (year >= CONSCIENCIA_NEGRA_SINCE) { + holidays.push({ name: "Dia da Consciência Negra", date: civilDate(year, 11, 20), type: "national" }); + } + + const easter = easterSunday(year); + + holidays.push({ + name: "Carnaval (terça-feira)", + date: date.addDays(easter, -47) ?? easter, + type: "optional", + }); + holidays.push({ + name: "Sexta-feira Santa", + date: date.addDays(easter, -2) ?? easter, + type: "national", + }); + holidays.push({ name: "Páscoa", date: easter, type: "religious" }); + holidays.push({ + name: "Corpus Christi", + date: date.addDays(easter, 60) ?? easter, + type: "optional", + }); + + return seq.sortStableBy(holidays, (holiday: Holiday): CivilDate => holiday.date); +} diff --git a/core/source/is-business-day.ts b/core/source/is-business-day.ts new file mode 100644 index 000000000..7bd0be15b --- /dev/null +++ b/core/source/is-business-day.ts @@ -0,0 +1,40 @@ +import { getHolidays } from "./get-holidays"; + +const SATURDAY = 6; +const SUNDAY = 7; + +/** + * Whether a date is a Brazilian business day (dia útil). + * + * A day is not a business day when it falls on a weekend, or when it is one of the holidays + * `getHolidays` lists for its year. `includeOptional` decides whether the ponto facultativo + * entries (Carnaval, Corpus Christi) count; the published package defaults it to `true`, and + * supplying that default is the DX's job. + * + * Only the years 1900 to 2099 are supported, the range the holiday rules are stated for. + */ +export function isBusinessDay(value: CivilDate, includeOptional: boolean): boolean { + const year = date.year(value); + + if (year < 1900 || year > 2099) { + return false; + } + + const weekday = date.dayOfWeek(value); + + if (weekday === SATURDAY || weekday === SUNDAY) { + return false; + } + + for (const holiday of getHolidays(year)) { + if (!includeOptional && holiday.type === "optional") { + continue; + } + + if (date.compare(holiday.date, value) === 0) { + return false; + } + } + + return true; +} diff --git a/core/source/is-valid-cnpj.ts b/core/source/is-valid-cnpj.ts new file mode 100644 index 000000000..5dee7898d --- /dev/null +++ b/core/source/is-valid-cnpj.ts @@ -0,0 +1,46 @@ +import { keepAlphanumeric, keepDigits } from "./lib/digits"; +import { hasLetter, hasValidCnpjChecksum, isRepeatedCnpj } from "./lib/cnpj"; + +/** Which CNPJ format to accept: the numeric one, or the alphanumeric one. */ +export type CnpjVersion = "1" | "2"; + +const CNPJ_FORMAT = + /^[0-9A-Z]{2}[\t-\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff.\-/]*[0-9A-Z]{3}[\t-\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff.\-/]*[0-9A-Z]{3}[\t-\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff.\-/]*[0-9A-Z]{4}[\t-\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff.\-/]*[0-9]{2}$/; + +const NUMERIC_CNPJ_FORMAT = + /^[0-9]{2}[\t-\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff.\-/]*[0-9]{3}[\t-\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff.\-/]*[0-9]{3}[\t-\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff.\-/]*[0-9]{4}[\t-\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff.\-/]*[0-9]{2}$/; + +const CNPJ_LENGTH = 14; + +/** + * Validates a CNPJ (Cadastro Nacional da Pessoa Jurídica), numeric or alphanumeric. + * + * Version `"2"` accepts the alphanumeric format as well; a value with no letters is always read + * as the numeric one, which is also where the reserved repeated numbers are rejected. Mapping a + * missing or unexpected `options.version` onto `"1"` is the DX's job. + */ +export function isValidCnpj(cnpj: string, version: CnpjVersion): boolean { + const trimmed = cnpj.trim(); + + if (version === "2") { + const cleaned = keepAlphanumeric(cnpj); + + if (hasLetter(cleaned) && cleaned.length === CNPJ_LENGTH) { + // Kept as `str.asciiUpper`: `.toUpperCase()` is only its ordinary spelling once the + // argument is proven ASCII (`requireAsciiCase`, docs/semantics.md §7.1), and `trimmed` is + // `cnpj.trim()` on the raw, unsanitized input — it carries no such proof, unlike + // `keepAlphanumeric`'s own `str.asciiUpper` call, which now reads `.toUpperCase()` + // because its argument is the result of `re.retain`, already typed `Ascii`. Proving + // `trimmed` ASCII first would mean restructuring this check, not respelling it. + return CNPJ_FORMAT.test(str.asciiUpper(trimmed)) && hasValidCnpjChecksum(cleaned); + } + } + + const numeric = keepDigits(cnpj); + + if (numeric.length !== CNPJ_LENGTH) { + return false; + } + + return NUMERIC_CNPJ_FORMAT.test(trimmed) && !isRepeatedCnpj(numeric) && hasValidCnpjChecksum(numeric); +} diff --git a/core/source/is-valid-cpf.ts b/core/source/is-valid-cpf.ts new file mode 100644 index 000000000..ebea9dde4 --- /dev/null +++ b/core/source/is-valid-cpf.ts @@ -0,0 +1,35 @@ +import { digitAt, keepDigits } from "./lib/digits"; +import { cpfCheckDigit, isRepeated } from "./lib/cpf"; + +/** + * The mask a CPF may be written with. The class spells out the 25 code points JavaScript's `\s` + * matches, because `\s` itself means a different set in Python and in Go. + */ +const CPF_FORMAT = + /^[0-9]{3}[\t-\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff.\-/]*[0-9]{3}[\t-\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff.\-/]*[0-9]{3}[\t-\r \u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000\ufeff.\-/]*[0-9]{2}$/; + +const CPF_LENGTH = 11; + +/** + * Validates a CPF (Cadastro de Pessoas Físicas). + * + * The core takes the value as written, accepting the usual mask characters; turning a host value + * into a string is the DX's job. + */ +export function isValidCpf(cpf: string): boolean { + if (!CPF_FORMAT.test(cpf.trim())) { + return false; + } + + const digits = keepDigits(cpf); + + if (digits.length !== CPF_LENGTH) { + return false; + } + + if (isRepeated(digits)) { + return false; + } + + return digitAt(digits, 9) === cpfCheckDigit(digits, 9) && digitAt(digits, 10) === cpfCheckDigit(digits, 10); +} diff --git a/core/source/lib/civil.ts b/core/source/lib/civil.ts new file mode 100644 index 000000000..9d1646d72 --- /dev/null +++ b/core/source/lib/civil.ts @@ -0,0 +1,17 @@ +/** + * Civil date construction in the source language. + * + * The core has no unchecked construction: `date.fromYmd` answers an absent value for a day that + * does not exist, and the checker insists on that case being handled. A literal date from a + * statute is always real, so this helper names the fallback once instead of repeating it at every + * call site. + */ + +/** A fixed day of a year, with the unreachable fallback named once. */ +export function civilDate( + year: IntRange<1900, 2099>, + month: IntRange<1, 12>, + day: IntRange<1, 31>, +): CivilDate { + return date.fromYmd(year, month, day) ?? date.clampEpochDays(0); +} diff --git a/core/source/lib/cnpj.ts b/core/source/lib/cnpj.ts new file mode 100644 index 000000000..4a376f41d --- /dev/null +++ b/core/source/lib/cnpj.ts @@ -0,0 +1,83 @@ +/** + * CNPJ rules shared by the CNPJ utilities. + * + * Both versions go through one check digit calculation: each character is read as its code point + * minus 48, which is the digit itself for `0` to `9` and the value the alphanumeric CNPJ assigns + * to `A` to `Z` (17 to 42), exactly as the Receita Federal manual specifies. + */ + +import { randomDigit } from "./random"; + +/** Exported for `generate-cnpj`, which needs the check digit of a base that has none yet. */ +export const FIRST_WEIGHTS = [5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]; + +export const SECOND_WEIGHTS = [6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]; + +/** The check digit of a CNPJ base, under the rule both versions share. */ +export function cnpjCheckDigit(cnpj: AsciiOf<14>, weights: List>): IntRange<0, 9> { + // The base is any 14 character ASCII value, so a character below '0' contributes a negative + // term; the check digit itself is still 0 to 9, which the return type proves. + let sum: IntRange<-6000, 10000> = 0; + + for (let index = 0; index < weights.length; index++) { + sum += (cnpj.charCodeAt(index) - 48) * weights[index]; + } + + const remainder = sum % 11; + + return remainder < 2 ? 0 : 11 - remainder; +} + +/** Whether both check digits of a 14 character CNPJ match its base. */ +export function hasValidCnpjChecksum(cnpj: AsciiOf<14>): boolean { + return ( + cnpj.charCodeAt(12) - 48 === cnpjCheckDigit(cnpj, FIRST_WEIGHTS) && + cnpj.charCodeAt(13) - 48 === cnpjCheckDigit(cnpj, SECOND_WEIGHTS) + ); +} + +/** + * Whether the value holds at least one upper cased ASCII letter. + * + * The scan reads positions rather than materializing the scalars, which the checked accessor + * makes safe without a proof about the length. + */ +export function hasLetter(value: Ascii): boolean { + for (let index = 0; index < value.length; index++) { + // `value[index]?.charCodeAt(0)` is the checked *numeric* accessor: `value.charCodeAt(i)` + // alone always answers `NaN` past the end, not `undefined`, but `value[index]` alone already + // answers `undefined` there, and `?.charCodeAt(0)` reads the one scalar's code point only + // when it is present — the ordinary spelling of `str.codeAtOpt`, which this unbounded loop + // needs because the index is never provably in range. + const point = value[index]?.charCodeAt(0) ?? 0; + + if (point >= 65 && point <= 90) { + return true; + } + } + + return false; +} + +/** + * A random numeric CNPJ base: an 8-digit root and a 4-digit branch, each digit drawn + * independently — matches the published `generateCnpj()` called with no branch, where an unset + * branch also draws those 4 digits at random. Twelve separate draws, not a loop, is what lets the + * result stay exactly 12 digits long. + */ +export function randomCnpjBase(): DigitsOf<12> { + return `${randomDigit()}${randomDigit()}${randomDigit()}${randomDigit()}${randomDigit()}${randomDigit()}${randomDigit()}${randomDigit()}${randomDigit()}${randomDigit()}${randomDigit()}${randomDigit()}`; +} + +/** Whether every character of a 14 character value is the same one. */ +export function isRepeatedCnpj(value: AsciiOf<14>): boolean { + const first = value.charCodeAt(0); + + for (let index = 1; index < 14; index++) { + if (value.charCodeAt(index) !== first) { + return false; + } + } + + return true; +} diff --git a/core/source/lib/cpf.ts b/core/source/lib/cpf.ts new file mode 100644 index 000000000..6042888ba --- /dev/null +++ b/core/source/lib/cpf.ts @@ -0,0 +1,45 @@ +/** + * CPF rules shared by the CPF utilities. + * + * Library code, not intrinsics: both functions are expressible in the subset, so every target + * gets the same implementation rather than a per-language shim. + */ + +import { digitAt } from "./digits"; +import { randomDigit } from "./random"; + +/** The check digit of a CPF base, under the Receita Federal rule (weights 10..2 and 11..2). */ +export function cpfCheckDigit(cpf: DigitsOf<11>, size: IntRange<9, 10>): IntRange<0, 9> { + let sum: IntRange<0, 1000> = 0; + + for (let index = 0; index < size; index++) { + sum += digitAt(cpf, index) * (size + 1 - index); + } + + const remainder = sum % 11; + + return remainder < 2 ? 0 : 11 - remainder; +} + +/** + * A random CPF base: 8 digits plus a região fiscal digit, each drawn independently — matches the + * published `generateCpf()` called with no state, where an unset state also draws that 9th digit + * at random. Nine separate draws, not a loop, is what lets the result stay exactly 9 digits long. + */ +export function randomCpfBase(): DigitsOf<9> { + return `${randomDigit()}${randomDigit()}${randomDigit()}${randomDigit()}${randomDigit()}${randomDigit()}${randomDigit()}${randomDigit()}${randomDigit()}`; +} + +/** Whether every scalar of the value is the same one, e.g. "00000000000". */ +export function isRepeated(value: DigitsOf<11>): boolean { + const first = value.charCodeAt(0); + + for (let index = 1; index < 11; index++) { + if (value.charCodeAt(index) !== first) { + return false; + } + } + + return true; +} + diff --git a/core/source/lib/digits.ts b/core/source/lib/digits.ts new file mode 100644 index 000000000..2fcc32cbf --- /dev/null +++ b/core/source/lib/digits.ts @@ -0,0 +1,39 @@ +/** + * Digit and ASCII helpers shared by the document utilities. + * + * `value.replace(/[^…]/g, "")` keeps only the scalars of a single character class, which is one + * pass in every target and refines the result to that class: the callers below get a `Digits` or + * an `Ascii` without a check of their own. + */ + +/** Keeps only the ASCII digits of a value, dropping every mask character. */ +export function keepDigits(value: string): Digits { + return value.replace(/[^0-9]/g, ""); +} + +/** Keeps only the ASCII digits and letters of a value, upper casing the letters. */ +export function keepAlphanumeric(value: string): Ascii { + return value.replace(/[^0-9A-Za-z]/g, "").toUpperCase(); +} + +/** The numeric value of one ASCII digit. */ +export function digitAt(value: Digits, index: number): IntRange<0, 9> { + return value.charCodeAt(index) - 48; +} + +/** + * Whether every scalar of the value is the same one, for whatever length the caller proved — + * `isRepeated` and `isRepeatedCnpj` do the same check for one specific length; this one serves a + * generator that has to run it on a base shorter than the document it is building. + */ +export function isRepeatedRun(value: Digits): boolean { + const first = value.charCodeAt(0); + + for (let index = 1; index < value.length; index++) { + if (value.charCodeAt(index) !== first) { + return false; + } + } + + return true; +} diff --git a/core/source/lib/easter.ts b/core/source/lib/easter.ts new file mode 100644 index 000000000..fa234d93d --- /dev/null +++ b/core/source/lib/easter.ts @@ -0,0 +1,36 @@ +import { civilDate } from "./civil"; + +/** + * Easter Sunday, with the Meeus/Jones/Butcher (anonymous Gregorian) algorithm. + * + * Library code, not an intrinsic: it is a handful of integer operations, so every target gets the + * same arithmetic instead of a per-language calendar call. Every intermediate is non-negative in + * the supported year range, so truncated division is the floor division the algorithm assumes. + */ + +/** The day of March (1 to 31) or April (32 to 56) Easter falls on, as a day-of-March offset. */ +export function easterDayOfMarch(year: IntRange<1583, 9999>): IntRange<22, 56> { + const a = year % 19; + const b = year / 100; + const c = year % 100; + const d = b / 4; + const e = b % 4; + const f = (b + 8) / 25; + const g = (b - f + 1) / 3; + const h = (19 * a + b - d - g + 15) % 30; + const i = c / 4; + const k = c % 4; + const l = (32 + 2 * e + 2 * i - h - k) % 7; + const m = (a + 11 * h + 22 * l) / 451; + const day = h + l - 7 * m + 114; + + // `day` counts from 1 March: 22 is 22 March, 56 is 25 April, the two ends of the Easter window. + return Math.min(Math.max(day % 31 + 1 + (day / 31 - 3) * 31, 22), 56); +} + +/** Easter Sunday of a year, as a civil date. */ +export function easterSunday(year: IntRange<1900, 2099>): CivilDate { + const dayOfMarch = easterDayOfMarch(year); + + return dayOfMarch <= 31 ? civilDate(year, 3, dayOfMarch) : civilDate(year, 4, dayOfMarch - 31); +} diff --git a/core/source/lib/format.ts b/core/source/lib/format.ts new file mode 100644 index 000000000..4377a743c --- /dev/null +++ b/core/source/lib/format.ts @@ -0,0 +1,80 @@ +/** + * Formatting shared by every `format*` utility: thousands grouping, and pattern formatting. + * + * A pattern is read one scalar at a time: `0` copies one input scalar, `*` hides one, and + * anything else is a separator, emitted only while the value still has scalars left. + * + * Both the pattern and the value are ASCII, so positions are O(1) everywhere and the result is + * built by concatenation rather than through a scalar list. + */ + +const SLOT = "0"; +const HIDDEN = "*"; + +/** How many digits a thousands group holds, in the pt-BR convention. */ +const GROUP_SIZE = 3; + +/** Groups the whole part with `.` every three digits, the pt-BR convention. */ +export function groupThousands(whole: Digits): Ascii { + let out: IntRange<0, 127>[] = []; + // Kept as `str.codePoints`: real TypeScript spreads a string into substrings (`string[]`), so + // `[...whole]` would not type-check against the numeric code points this loop pushes below, + // even though the engine's own checker treats the two spellings as identical Core. + const scalars = str.codePoints(whole); + + for (let index = 0; index < scalars.length; index++) { + if (index > 0 && (scalars.length - index) % GROUP_SIZE === 0) { + out.push(46); + } + + out.push(scalars[index] ?? 48); + } + + return str.fromCodePoints(out); +} + +/** How many scalars of the value a pattern consumes. */ +export function patternSlots(pattern: Ascii): IntRange<0, 2147483647> { + let slots: IntRange<0, 2147483647> = 0; + + for (let index = 0; index < pattern.length; index++) { + // `pattern[index] ?? ""`: `pattern` is always a fixed-length literal at its call sites + // (`PATTERN`, `OBFUSCATED_PATTERN`), so specialization proves this loop's index in range — + // but the `??` is the author's own statement that the absent case is wanted regardless, so + // it picks the checked `str.charAtOpt` unconditionally (see the `logical` handling of `??` + // on a bracket index), the same Core the namespace form always produced here. + const symbol = pattern[index] ?? ""; + + if (symbol === SLOT || symbol === HIDDEN) { + slots += 1; + } + } + + return slots; +} + +/** Formats a value against a pattern, optionally left padding it with zeros first. */ +export function formatWithPattern(value: Ascii, pattern: Ascii, pad: boolean): Ascii { + const padded = pad ? value.padStart(patternSlots(pattern), SLOT) : value; + + let out: Ascii = ""; + let taken: Int = 0; + + // `pattern[index] ?? ""` here for the same reason as `patternSlots` above. + for (let index = 0; index < pattern.length; index++) { + const symbol = pattern[index] ?? ""; + + if (symbol === SLOT || symbol === HIDDEN) { + if (taken >= padded.length) { + return out; + } + + out = out + (symbol === HIDDEN ? HIDDEN : padded[taken] ?? ""); + taken += 1; + } else if (taken < padded.length) { + out = out + symbol; + } + } + + return out; +} diff --git a/core/source/lib/json.ts b/core/source/lib/json.ts new file mode 100644 index 000000000..d3158f145 --- /dev/null +++ b/core/source/lib/json.ts @@ -0,0 +1,147 @@ +/** + * A JSON string-field reader, written in the source language. + * + * It is library code rather than an intrinsic for the usual reason: it is expressible in the + * subset, so every target runs the same scanner instead of three different JSON libraries with + * three different edge cases. It reads exactly what the core needs — the string value of a + * top-level field — and nothing else. + */ + +const QUOTE = 34; +const BACKSLASH = 92; +const COLON = 58; +const SPACE = 32; +const TAB = 9; +const NEWLINE = 10; +const RETURN = 13; + +/** Whether `needle` occurs in `points` at `start`. */ +function matchesAt(points: List, needle: List, start: number): boolean { + for (let offset = 0; offset < needle.length; offset++) { + // `needle[offset] ?? -2`: every call site passes a fixed-length literal key (`"cep"`, + // `"uf"`, …), so specialization proves this loop's index in range for `needle` — but the + // `??` picks the checked `seq.at` regardless (see the `logical` handling of `??` on a + // bracket index), the same Core the namespace form always produced here. `points[start + + // offset]` has no such proof either way (the body is any HTTP response), so it was already + // the checked form on its own. + if ((points[start + offset] ?? -1) !== (needle[offset] ?? -2)) { + return false; + } + } + + return true; +} + +/** Whether a code point is JSON whitespace. */ +function isSpace(point: number): boolean { + return point === SPACE || point === TAB || point === NEWLINE || point === RETURN; +} + +/** The hexadecimal value of four scalars, for a `\uXXXX` escape. */ +function hexValue(points: List, start: number): IntRange<0, 65535> { + let value: IntRange<0, 1114111> = 0; + + for (let offset = 0; offset < 4; offset++) { + const point = points[start + offset] ?? 48; + let digit: IntRange<0, 15> = 0; + + if (point >= 48 && point <= 57) { + digit = point - 48; + } else if (point >= 97 && point <= 102) { + digit = point - 87; + } else if (point >= 65 && point <= 70) { + digit = point - 55; + } + + value = value * 16 + digit; + } + + return Math.min(value, 65535); +} + +/** + * The string value of a top-level JSON field, or absent when the field is missing or is not a + * string. Escapes are decoded; a surrogate pair is left as its two escaped halves, which no CEP + * provider emits. + */ +export function jsonStringField(body: string, key: Ascii): string | undefined { + // Kept as `str.codePoints`: real TypeScript spreads a string into substrings (`string[]`), not + // the numeric code points every scan below compares against, even though the engine's own + // checker treats `[...s]` and this call as identical Core. + const points = str.codePoints(body); + const needle = str.codePoints(`"${key}"`); + + for (let index = 0; index < points.length; index++) { + if (!matchesAt(points, needle, index)) { + continue; + } + + // Declared over the platform domain: the cursor walks the whole body, and every step is a + // small constant, which is what lets the checker keep it inside that domain. + let cursor: Int = index + needle.length; + + for (let skip = 0; skip < 8; skip++) { + if (isSpace(points[cursor] ?? 0)) { + cursor += 1; + } + } + + if ((points[cursor] ?? 0) !== COLON) { + continue; + } + + cursor += 1; + + for (let skip = 0; skip < 8; skip++) { + if (isSpace(points[cursor] ?? 0)) { + cursor += 1; + } + } + + if ((points[cursor] ?? 0) !== QUOTE) { + continue; + } + + cursor += 1; + + let out: IntRange<0, 1114111>[] = []; + + for (let step = 0; step < points.length; step++) { + const point = points[cursor] ?? -1; + + if (point === -1 || point === QUOTE) { + return str.fromCodePoints(out); + } + + if (point === BACKSLASH) { + const escaped = points[cursor + 1] ?? -1; + + if (escaped === 110) { + out.push(NEWLINE); + cursor += 2; + } else if (escaped === 116) { + out.push(TAB); + cursor += 2; + } else if (escaped === 114) { + out.push(RETURN); + cursor += 2; + } else if (escaped === 117) { + out.push(hexValue(points, cursor + 2)); + cursor += 6; + } else if (escaped >= 0) { + out.push(escaped); + cursor += 2; + } else { + cursor += 1; + } + } else { + out.push(point); + cursor += 1; + } + } + + return str.fromCodePoints(out); + } + + return undefined; +} diff --git a/core/source/lib/random.ts b/core/source/lib/random.ts new file mode 100644 index 000000000..567173a3c --- /dev/null +++ b/core/source/lib/random.ts @@ -0,0 +1,45 @@ +/** + * Random, derived from the one intrinsic, `random.nextU32`. + * + * Everything past that single draw is written here, in source: a range or a shuffle would + * otherwise have to match, bit for bit, across three unrelated standard libraries, which only a + * shared implementation can guarantee. + */ + +/** One past the highest value `random.nextU32` can answer. */ +const U32_SPAN = 4294967296; + +/** + * How many draws `randomBelow` allows itself before it falls back to a biased `% bound`. The + * subset has no unbounded `while`, so the search has to be a counted `for`. The worst case, a + * bound just over half of 2^32, rejects just under half of every draw, so 32 attempts leave the + * fallback below reached with probability under 2^-32. + */ +const MAX_ATTEMPTS = 32; + +/** + * A uniform integer in `[0, bound)`, by rejection sampling rather than `% bound`: the modulo of a + * fixed-width draw is biased whenever `bound` does not divide 2^32 evenly, and that bias would + * have to match, digit for digit, across three unrelated standard libraries to stay invisible. + * Rejecting the biased tail of the draw removes it instead. + */ +export function randomBelow(bound: IntRange<1, 4294967296>): IntRange<0, 4294967295> { + const limit = U32_SPAN - (U32_SPAN % bound); + + for (let attempt = 0; attempt < MAX_ATTEMPTS; attempt++) { + const draw = random.nextU32(); + + if (draw < limit) { + return draw % bound; + } + } + + // Every attempt above landed in the biased tail, a probability under 2^-32; `% bound` here is + // the very bias `randomBelow` exists to avoid everywhere else, taken as a documented fallback. + return random.nextU32() % bound; +} + +/** One random ASCII digit. */ +export function randomDigit(): Digits { + return String(randomBelow(10)); +} diff --git a/core/tsconfig.json b/core/tsconfig.json new file mode 100644 index 000000000..2c1b3f61b --- /dev/null +++ b/core/tsconfig.json @@ -0,0 +1,14 @@ +{ + "compilerOptions": { + "lib": ["ESNext"], + "target": "ESNext", + "module": "ESNext", + "moduleResolution": "bundler", + "allowImportingTsExtensions": true, + "noEmit": true, + "strict": true, + "skipLibCheck": true, + "types": [] + }, + "include": ["source/**/*.ts", "../engine/prelude/index.d.ts"] +} diff --git a/engine/.gitignore b/engine/.gitignore new file mode 100644 index 000000000..9209ef5bf --- /dev/null +++ b/engine/.gitignore @@ -0,0 +1,2 @@ +node_modules +out diff --git a/engine/README.md b/engine/README.md new file mode 100644 index 000000000..2ee3b8454 --- /dev/null +++ b/engine/README.md @@ -0,0 +1,89 @@ +# Logic engine + +Write a library's logic **once**, in a restricted, semantically typed subset of TypeScript, and +generate a native, idiomatic implementation for TypeScript, Python and Go. + +No shared runtime package. No WASM, no FFI, no bridge. No interpreter at run time. No external +dependencies in the generated code. + +```sh +npm install # one runtime dependency, the parser; prettier and tsc for development +npm test # the engine's own suite +node src/cli.ts build --project ../core +node scripts/verify.ts ../core +``` + +## What it is + +A semantic compiler for a constrained library language — not a universal transpiler. + +``` +source (restricted TypeScript) + │ frontend: parse, resolve, reject what is outside the subset + ▼ +Semantic HIR ── the frontend contract + │ checker: types, refinements, ranges, effects → Core + ▼ +Core IR ── what the program means, in no particular language + │ link, comptime, optimize, capability threading + ▼ +per target: lowering selection → Target AST → printer → formatter + ▼ +TypeScript · Python · Go +``` + +The engine generates the **core**, never the public API. The handwritten DX in each language keeps +its own coercion, defaults and naming, and calls a core whose signatures are stable and versioned. + +## Why the types are the point + +`Int` carries a proven range. `String` carries a character class and a length range. Those are not +decoration: they are what makes a native lowering *provably* equivalent. + +```ts +str.compare(left, right) +``` + +- Python: `<` compares code points — the Core's order — so the native comparison is selected with + no precondition. +- Go: `strings.Compare` compares UTF-8 bytes, which is the same order, so native again. +- TypeScript: `<` compares **UTF-16 code units**, which sorts an astral scalar below U+E000. The + native comparison is selected only when both sides are proven ASCII; otherwise the portable + implementation from the engine's own source-language standard library is used. + +One line of source, three correct answers, and `out//LOWERING.md` says which rule decided +each one. + +## What is generated + +For each target: one module per source module, a `capabilities` file with the default environment +built from that language's standard library, an `errors` file, `LOWERING.md` (every non-trivial +selection and why), `API.json` (the published core signatures) and `SOURCEMAP.json` (each +generated function back to its source span). Every file carries a provenance header. + +Each target is also generated in a `--no-idioms` mode, and conformance runs both, which is what +proves idiom selection preserves meaning. + +## Using it in another project + +The engine knows nothing about any particular library; `examples/generic` is a project with no +relation to the one that motivated it, and `tests/example.spec.ts` keeps it that way. + +``` +my-project/ + engine.config.json { "name", "sourceRoot", "out", "targets" } + source/ + my-utility.ts one exported function per file at the root + lib/… library code, specialized per call site + conformance/… optional: cases and a runner +``` + +```sh +node /src/cli.ts build --project my-project +node /scripts/verify.ts my-project +``` + +## Documentation + +[`docs/`](docs) — the specification, the generated intrinsic reference, how to add a utility or a +target, the per-target notes, the architectural decisions, and the measured metrics. diff --git a/engine/docs/README.md b/engine/docs/README.md new file mode 100644 index 000000000..78b7d1ae7 --- /dev/null +++ b/engine/docs/README.md @@ -0,0 +1,14 @@ +# Engine documentation + +- [semantics.md](semantics.md) — the specification: types, refinements, effects, the subset, and + every rule that exists because two languages disagree. +- [intrinsics.md](intrinsics.md) — generated from the registry: every operation the Core can + express. +- [adding-a-utility.md](adding-a-utility.md) — the workflow, including the habits that make the + checker's job possible. +- [adding-a-target.md](adding-a-target.md) — what a backend is made of. +- [fuzzing.md](fuzzing.md) — the random program generator: what it generates, the two comparisons + it runs, and the seed-and-shrink story. +- [targets/](targets) — representation and notable lowerings per target, plus the Rust sketch. +- [decisions/](decisions) — the architectural decisions and the evidence behind them. +- [progress.md](progress.md) — milestone status and the measured metrics. diff --git a/engine/docs/adding-a-target.md b/engine/docs/adding-a-target.md new file mode 100644 index 000000000..ecd67bf4d --- /dev/null +++ b/engine/docs/adding-a-target.md @@ -0,0 +1,93 @@ +# Adding a target + +A target is a capability table, a printer, four structural flags and a support file. There is no +new tree and no new pass. + +## 1. Decide the representation + +Fill in the table in `docs/targets/.md` before writing code: what each semantic type +becomes, and what each choice assumes. The two that matter most are the integer representation +(what the language's default integer covers, and what a wider range forces) and the string +representation (what "index" means, and whether comparison is code point order). + +## 2. Write the spec + +```ts +export const LANGUAGE_SPEC: TargetSpec = { + name: "language", + table: new LoweringTable(LANGUAGE_CANDIDATES), + naming: { func, value, field, type, module }, + loopCombinators: new Set(["seq.fold"]), // what reads better as a loop here + statementTernary: false, // true when there is no conditional expression + errorsAsValues: false, // true when a failure is a second return value + asyncColouring: false, // true when reaching Http makes a function async + envType: { kind: "Record", name: "Capabilities" }, +}; +``` + +## 3. Write the capability table + +One entry per intrinsic the target can lower, each with: + +- `impl`: `native`, `library` (the language's standard library) or `portable` (a call into the + engine's source-language standard library); +- `requires`: the facts the lowering needs, and `because`: why it needs them, which is printed in + `LOWERING.md`; +- `cost`: allocations and time complexity, which is how selection ranks candidates; +- `emit`: the Target AST fragment. + +A missing entry is a compile error naming the operation and the argument types, so a partial table +is a usable table: the operations a project does not use never have to be written. + +Where the language's own function is only conditionally equivalent, say so in `requires` rather +than in a comment. `String#length` counting UTF-16 code units is a precondition, not a footnote. + +### `emit` returning `raw` hides everything inside it + +`emit` may return a `raw` node carrying printed text, and every target does for the shapes that +have no node kind. It costs something: once an argument has been printed into that text, it is a +string, and no later pass over the Target AST can see it. `hoistConstantTables` lifts a constant +list out of a function body by walking that AST, so a table that reached a `raw` emission stays +inline — which is why the generated Go builds its weight table on every call in `generate-cnpj` +while the generated TypeScript, whose lowering of the same call keeps the argument as a node, +lifts it to module scope. + +Prefer a structured node with the arguments as children (`call`, `method`, `binary`, `ternary`) +and keep `raw` for leaves. Where the target really needs text around an argument, know that +anything inside it is final. + +The same applies in reverse to work a target wants done once: a pattern, a table, a lookup built +from a compile-time constant does not belong in the call path. Go compiles its regexes into +package level `var`s, Python into module level `re.compile`, and Rust turns each into a `static` +or a dedicated scanner, all decided at generation time. Each of those was a measured defect before +it was a rule — the Go one cost 79.6x the price of the match it was performing. + +## 4. Write the printer and the support file + +The printer renders the Target AST. The support file holds what the engine generates rather than +depends on: the capability interface and its default implementation from the language's standard +library, plus whatever generic helpers the capability table names. + +## 5. Prove it + +Add the backend to `src/cli.ts`, then: + +```sh +node scripts/verify.ts ../core +``` + +`verify` generates in both idiom modes, regenerates and diffs for determinism, runs the language's +own linters over the output, and runs the differential conformance harness: every case, through +the generated driver, compared against the reference interpreter. A target is done when that is +green and a fluent reader would accept the golden files. + +Then run the random program generator against the new target on its own, alongside the hand-written +cases: + +```sh +node scripts/fuzz.ts full --seed 20260921 --count 300 --targets +``` + +`fuzz full` does not know which target it is comparing — it generates cases and compares answers, +the same way for every target — so this gets the new backend the same generated coverage the first +four had, without writing a second harness. See [fuzzing.md](fuzzing.md). diff --git a/engine/docs/adding-a-utility.md b/engine/docs/adding-a-utility.md new file mode 100644 index 000000000..2dddbd663 --- /dev/null +++ b/engine/docs/adding-a-utility.md @@ -0,0 +1,92 @@ +# Adding a utility + +A utility is one exported function in one file at the source root. Everything it needs that is not +an intrinsic goes under `source/lib/`. + +A plain `function` declared alongside a utility, with no `export`, stays private in every generated +target — `export` in TypeScript, a leading underscore plus `__all__` in Python, a lower-case +initial in Go, `pub`/`pub(crate)`/nothing in Rust, decided per function from whether *its own* +source module exported it, not from whether the utility that happens to call it did. Write such a +helper the same way you would in the published package: unexported, because it is not part of what +this file offers the rest of the project. + +A utility whose effects reach `Http`, `Clock` or `Random` never takes a capability parameter in its +own published signature — see `docs/semantics.md` §4.1 and +[ADR 0011](decisions/0011-public-entry-points-vs-capabilities.md). Nothing about writing the +utility changes for this: call `http.request(…)`, `clock.now()` or `random.nextU32()` exactly as +`core/source/get-address-info-by-cep.ts` and `core/source/generate-cpf.ts` do, and never mention an +environment. The split into a public wrapper and a capability-taking seam happens after checking, +from the effects the checker already inferred; there is nothing to opt into or write differently. + +## 1. Measure the behavior first + +Write down what the existing implementation does, including the parts that look like accidents: +the exact separator, whether a mask character is allowed between groups, what an empty string +answers. `core/docs/contracts.md` is where those measurements live. Never invent behavior. + +Split the measurement in two: what the **core** does, and what the **DX** does (coercing a number +to a string, defaulting an option, converting a host `Date`). The core takes already-normalized +values. + +## 2. Write the source + +```ts +// source/is-valid-example.ts +import { digitAt, keepDigits } from "./lib/digits"; + +const FORMAT = /^[0-9]{5}$/; + +/** + * One sentence about what this validates, and one about what the DX still owes it. + */ +export function isValidExample(value: string): boolean { + if (!re.test(FORMAT, str.trim(value))) { + return false; + } + + const digits = keepDigits(value); + + if (digits.length !== 5) { + return false; + } + + return digitAt(digits, 4) === 7; +} +``` + +Three habits that make the checker's job possible: + +- **guard, don't assert.** A guard (`if (value.length !== 5) return false;`) narrows the type; a + predicate function does not, because the checker reads a guard, not a called function. +- **annotate an accumulator** with the range you expect (`let sum: IntRange<0, 1000> = 0;`). The + checker proves the real range and tells you when it disagrees. +- **let a helper be a helper.** Library code is specialized per call site, so a helper that takes + `Digits` will be checked against the exact length its caller proved. + +## 3. Check, generate, verify + +```sh +npm run check # types, ranges, refinements, effects +npm run dump # the annotated Core, for reviewing what was proven +npm run build # every target, in both idiom modes +npm run verify # everything, including conformance +``` + +If the checker refuses something, read the suggestion: most rejections name the construct to write +instead. If it refuses something that is genuinely expressible, that is a bug or a missing +refinement — open it as one rather than weakening the checker. + +## 4. Add conformance cases + +In `conformance/cases.ts`, add the vectors that pin the documented edges, plus seeded random +inputs over the shape the DX really passes. In `conformance/run.ts`, add the published +implementation to `REFERENCE`, with the DX conversion written explicitly — that conversion is part +of the contract you measured. + +## 5. Review the generated code + +Read `out/typescript/.ts`, `out/python/.py` and `out/go/.go`. They +should look like code a fluent author would have written. If one of them does not, the answer is +usually a missing candidate in that target's capability table, not a change to the source. + +`out//LOWERING.md` says which lowering was selected for every operation and why. diff --git a/engine/docs/bindings-or-generated-source.md b/engine/docs/bindings-or-generated-source.md new file mode 100644 index 000000000..d98837106 --- /dev/null +++ b/engine/docs/bindings-or-generated-source.md @@ -0,0 +1,410 @@ + + +# One core, every language: bindings or generated source? + +Instead of generating **source** for each language, could Brazilian Utils build the logic once +— in Rust, in C, in WebAssembly — and give every package a **binding** to it? Is there an off +the shelf tool for that? And what does it cost at run time? + +What is being shared either way is the **utility's implementation**, not the published package. +Every ecosystem keeps writing its own DX by hand: its naming, its option objects, its types, its +docs. That matters for the answer, because a hand-written binding is then no more hand-written +code than the wrapper it sits behind. + +> **Revised.** The first version of this document concluded that generated source beat every +> binding. It had only measured the bindings a script reaches for — `ctypes`, Fiddle, a wasm +> runtime — and not the ones a package ships. A CPython extension is 16× cheaper than `ctypes` +> and P/Invoke costs about a nanosecond, which changes the answer for four of the seven +> ecosystems. [§2](#2-the-boundary-costs-more-than-the-work--but-only-over-the-wrong-boundary) +> has the corrected numbers and what they turn on. + +Everything below was measured on this branch; [Reproducing](#reproducing) says how to get the +harness back and re-run it. + +--- + +## The answer + +**It depends on the ecosystem, and the deciding factor is not speed.** + +The first pass of this document concluded that generated source beat every binding. That +conclusion was drawn from the wrong measurements: it compared generated source against the +bindings a _script_ reaches for — `ctypes`, Fiddle, a wasm runtime — and never against the +bindings a _package_ ships. Those are not the same thing, and they are not close: a CPython +extension module costs 27 ns per call where `ctypes` costs 433 ns, and P/Invoke with the GC +transition suppressed costs about 1 ns. + +Measured properly, with the core called the way each ecosystem actually calls native code: + +| Ecosystem | Best per call | Against its own generated source | Boundary cost | +| ---------- | --------------------------- | -------------------------------- | ------------- | +| **C#** | **76 ns** P/Invoke | 4.3× faster than handwritten | **~1 ns** | +| **Python** | **65 ns** CPython extension | **21× faster** | 27 ns | +| **Ruby** | **109 ns** C extension | **18× faster** | 65 ns | +| **Java** | **127 ns** Panama FFM | 3.9× faster | 62 ns | +| **Go** | 119 ns cgo | 1.6× faster | 58 ns | +| **Node** | 257 ns generated source | binding not available | — | +| **Rust** | 47 ns | it _is_ the core | 0 | + +So a C ABI really is close to free in C#, cheap in Python, and about 60 ns everywhere else — +against a validator body of 47 ns. That is the correction. + +What it does **not** change is the recommendation for the npm package, and that turns out to +decide the shape of the whole thing: + +- **JavaScript cannot take a binary core**, because the package is tree shakeable and a wasm + module is not. One utility as generated source is 509 bytes and disappears when unused; the + wasm core is 9,291 bytes, indivisible, and asynchronous to start. This is not a preference, + it is a requirement, so JavaScript gets generated source whatever the other ecosystems do. +- **Go pays for cgo in something other than nanoseconds.** It loses cross compilation, static + binaries and `CGO_ENABLED=0` builds, for a 74 ns gain. Not worth it. +- **Everywhere else the binding is the better answer**, and the ship cost — a native artifact + per platform × arch — is a cost those ecosystems already pay routinely. + +Which points at a hybrid rather than a winner, and a smaller tool than either option alone: +write the utility once, emit **TypeScript** for npm and **Rust** for the shared core, and let +Python, Ruby, C#, Java and Erlang bind to the compiled core with a hand-written binding. Go +takes generated source. The full recommendation is [at the end](#recommendation). + +## How it was measured + +One function — `isValidCpf` under a single shared profile — over a fixed corpus of 1000 +inputs: valid CPFs bare and masked, near misses with one digit flipped, +junk, Unicode whitespace, NBSP, ZWNBSP, an emoji, and CPFs written in full width and Arabic-Indic +digits. Each arm reports the best nanoseconds per call over 7 repetitions after warmup, plus how +many inputs it called valid — an arm cannot win by doing less work. + +Every arm computes the same thing, and the source level arms are generated from the **same +profile** — the harness overrides each package's adopted profile for the run — so the comparison +is not measuring one package's laxer contract against another's. + +Machine: Intel Xeon @ 2.80GHz, 4 vCPU, Linux. Node 22.22.2, Python 3.11.15, Ruby 3.3.6, Go 1.24.7, +rustc 1.94.1, OpenJDK 21.0.10, wasmtime-py 48.0.0, wasmtime gem 48.0.1, wazero 1.9.0, UniFFI 0.29. + +### Arms + +| Arm | What it is | +| --------------- | ------------------------------------------------------------------------- | +| `handwritten` | Idiomatic code a contributor would write in that language, regex and all | +| `generated` | The emitter's output for the same spec | +| `wasm` | The Rust core compiled to `wasm32-unknown-unknown`, one call per input | +| `wasm-batch` | Same core, the whole corpus in one call | +| `wasm-callonly` | The boundary alone: same pointer, same length, no marshalling | +| `ffi` / `cgo` | The same core as a `cdylib`, over ctypes / Fiddle / cgo | +| `uniffi` | The same logic behind UniFFI generated Python bindings | +| `cext` | The same core as a CPython extension module / a Ruby C extension | +| `ffm` | The same core over Java's Foreign Function & Memory API, trivial downcall | +| `ffi` (C#) | The same core over P/Invoke with `[SuppressGCTransition]` | +| `core-direct` | The same core called from C: the floor, the work with no boundary at all | + +--- + +## Results + +Lower is better. `× hand` is the ratio to that language's handwritten arm, so **below 1.00 is +faster than handwritten**. + +| Language | Arm | ns/op | × hand | valid | +| -------- | ------------------ | ---------: | -------: | ------: | +| C | **core-direct** | **47.0** | — | 337 | +| Node | handwritten | 386.2 | 1.00 | 337 | +| Node | **generated** | **256.5** | **0.66** | 337 | +| Node | wasm | 224.6 | 0.58 | 337 | +| Node | wasm-batch | 240.5 | 0.62 | 337 | +| Python | handwritten | 2460.6 | 1.00 | **333** | +| Python | **generated** | **1366.0** | **0.56** | 337 | +| Python | wasm | 38039.3 | 15.46 | 337 | +| Python | wasm-callonly | 30058.9 | 12.22 | — | +| Python | wasm-batch | 306.3 | 0.12 | 337 | +| Python | ffi (ctypes) | 535.3 | 0.22 | 337 | +| Python | ffi-batch | 239.9 | 0.10 | 337 | +| Python | **uniffi** | **9064.2** | **3.68** | 337 | +| Python | uniffi-batch | 6195.8 | 2.52 | 337 | +| Python | **cext** | **65.4** | **0.03** | 337 | +| Python | cext-batch | 235.9 | 0.10 | 337 | +| Ruby | handwritten | 3675.8 | 1.00 | **329** | +| Ruby | **generated** | **2020.1** | **0.55** | 337 | +| Ruby | wasm | 729.9 | 0.20 | 337 | +| Ruby | wasm-batch | 430.0 | 0.12 | 337 | +| Ruby | ffi (Fiddle) | 1662.1 | 0.45 | 337 | +| Ruby | **cext** | **109.3** | **0.03** | 337 | +| Go | handwritten | 446.2 | 1.00 | **334** | +| Go | **generated** | **193.1** | **0.43** | 337 | +| Go | wasm (wazero) | 189.1 | 0.42 | 337 | +| Go | wasm-batch | 104.5 | 0.23 | 337 | +| Go | cgo | 119.1 | 0.27 | 337 | +| Go | cgo-batch | 52.6 | 0.12 | 337 | +| Java | handwritten | 1365.3 | 1.00 | 337 | +| Java | **generated** | **493.6** | **0.36** | 337 | +| Java | **ffm (Panama)** | **127.3** | **0.09** | 337 | +| C# | handwritten | 330.2 | 1.00 | 337 | +| C# | **ffi (P/Invoke)** | **76.4** | **0.23** | 337 | + +--- + +## Seven findings + +### 1. Generated source is faster than handwritten, in every language measured + +0.36× in Java, 0.43× in Go, 0.56× in Python, 0.55× in Ruby, 0.66× in Node. Not because the generator is +clever, but because a human writing these by hand reaches for the regex engine (`\d{3}[\s.-]*...`) +while the emitter knows the exact character sets and can pick whichever construct is cheapest in +that language. + +This did **not** hold on the first measurement. The emitters used to walk the string code point by +code point in every language, which is right for Go and Java and terrible for Python and Ruby +(6.5 µs and 15.1 µs per call — 2.6× and 4.1× _slower_ than handwritten). Two rules fixed it, and +both are in this branch: + +- **Collapse `guard-shape` + `sanitize` into one anchored regex with capture groups.** The + character classes are written out code point by code point (`[\u0009\u000a…\u2000-\u200a…]`), so + the semantics stay exactly the spec's — `\s` would not — while the matching happens in C. The + digits come out of the capture groups, so there is no second pass over the string and no + intermediate allocation. +- **Read digits as bytes in the check digit loop** (`ord(char) - 48`, `getbyte(i) - 48`) instead of + `int(char)` / `to_i`, and unroll the verification instead of iterating positions with a closure. + +Python went 6485 → 1366 ns/op; Ruby 15051 → 2020. Conformance stayed at 2229/2229 for both. + +**This is the load bearing result for the whole idea**: the emitter is allowed to know things +about its target language, and the moment it does, generated code stops being a compromise. + +The `generated` arms came from the first prototype, `spec/codegen`, which described a utility as +JSON and emitted six languages from it. That prototype is retired — writing a utility as data +stopped scaling at the third one — and [`spec/bridge`](bridge/README.md) replaced it, so the +numbers above are a recording rather than something the current tree re-runs. What carried over +is the rule they established, and every emitter in `spec/bridge` is written to it. + +### 2. The boundary costs more than the work — but only over the wrong boundary + +The validator body is 47 ns, measured by calling the same shared library from C +(`core-direct`). Against that, here is every boundary measured, isolated: the same pointer and +length every call, no marshalling, minus the 40 ns the same fixed input costs in C. + +| Host | Mechanism | Boundary cost per call | +| ------------- | ---------------------------------- | ---------------------- | +| C# → C | P/Invoke, `[SuppressGCTransition]` | **~1 ns** | +| Python → C | CPython extension module | **27 ns** | +| Go → C | cgo | 58 ns | +| Java → C | Panama FFM, trivial downcall | 62 ns | +| Ruby → C | C extension | 65 ns | +| Go → wasm | wazero | 64 ns | +| Python → C | ctypes | 433 ns | +| Ruby → wasm | wasmtime gem | 482 ns | +| Ruby → C | Fiddle | 1,479 ns | +| Python → wasm | wasmtime-py 48 | **30,019 ns** | + +The top half and the bottom half are the same idea over different plumbing, and they differ by +three orders of magnitude. The first pass of this document measured only the bottom half and +drew a general conclusion from it; that was wrong. + +Three details decide which half a binding lands in, and all three are easy to get wrong: + +- **C#** needs `[SuppressGCTransition]`. Without it the call still costs only 42 ns rather than + 41, because .NET's transition is already cheap — this is the one runtime where the naive + version is fine. +- **Java** needs the `MethodHandle` to be `static final` and the off-heap buffer to come from + `Arena.global()`. With a shared arena and an instance handle the same call costs 84 ns more, + because the JIT cannot fold the downcall stub and pays a liveness check per call. + `Linker.Option.isTrivial()` itself is worth about 1 ns here. +- **Python and Ruby** need a real extension module. `ctypes` is 16× more expensive than a + CPython extension; Fiddle is 23× more expensive than a Ruby C extension. + +The corollary about batching still holds, and is now less interesting: amortising one crossing +over 1000 inputs helps the expensive boundaries and _hurts_ the cheap ones. Python's C +extension is 65 ns per call and 236 ns per call in batch, because packing the buffer in Python +costs more than the calls it saves. + +### 3. Binding quality varies by 300× between runtimes of the same technology + +Same `.wasm` file, same core, same machine: + +- wazero (pure Go): **104 ns** per call +- wasmtime gem (Rust native extension): **521 ns** +- wasmtime-py 48 (ctypes over the C API): **30 059 ns** + +wasmtime-py is not slow at running wasm; it is slow at _being called_, because every invocation +builds `Val` arrays through ctypes. Poking the linear memory through its raw pointer instead of +`Memory.write` only takes it from 38.0 µs to 32.4 µs — the cost is the call, not the copy. + +So "we ship a wasm core" is not one decision with one performance profile. It is a different +decision per package, and in Python today it means "and also batch everything". + +### 4. The off the shelf tool is slower than writing Python + +UniFFI is the tool everyone recommends for this (Mozilla ships Firefox features with it). Its +generated Python bindings cost **9064 ns per call — 3.7× slower than just implementing the +validator in Python**, and 17× slower than the same `.so` called through hand written ctypes. + +The generated call path explains it: `check_lower` validates the string, `lower` allocates a +`RustBuffer` and copies into it, a `_UniffiRustCallStatus` struct is built, the call goes through +ctypes, the status is checked, the result is lifted. Six Python level operations wrapped around +100 ns of work. Even `uniffi-batch` (a `Vec` in, a `Vec` out) lands at 6196 ns/op, +still slower than plain Python, because the sequence has to be serialised into a `RustBuffer`. + +UniFFI is built for coarse grained APIs — "sync this database", "decrypt this blob" — where a +microsecond of glue is irrelevant. Brazilian Utils is the opposite: a hundred tiny pure functions. + +### 5. The JS package would pay the most and gain the least, which settles it + +| | Generated source | Wasm core | +| -------------------------------------------- | --------------------------- | ----------------------------------------------------------------------- | +| `isValidCpf` single import bundle, minified | **509 bytes** | 9291 bytes (4344 gzipped) for the core with **one** utility | +| Tree shaking | per function, as today | none: the module is one indivisible blob | +| Bundling, Deno, Bun, browsers, edge runtimes | works everywhere, no config | needs a loader, async instantiation, and a bundler that handles `.wasm` | +| `sideEffects: false` and zero dependencies | preserved | gone | + +An 18× size increase for one function, no tree shaking, and asynchronous initialisation, in +exchange for 160 ns. The npm package's selling points are exactly what a wasm core takes away. + +Cold start, for the record: instantiating the 9 KB module costs 0.025 ms in Node, 3.5 ms in +Python, 3.8 ms in Ruby — fine for a server, not free for a CLI or a lambda. + +### 6. A binding is cheap; a binary is not + +Every number above is per call. The cost that decides the question is per release. + +Generated source ships as source: it is reviewed in the package's own repository, `git diff` +shows what changed, a stack trace points at a line, and there is no build matrix at all. A +shared core ships as an artifact per platform × arch per ecosystem, and needs a release +pipeline that produces them, a fallback for platforms nobody built, and an ABI that the +package and the artifact both agree on. + +Two ecosystems make that trade badly: + +- **JavaScript**, because of §5: no tree shaking, 18× the bytes, asynchronous startup. +- **Go**, because cgo costs cross compilation, static binaries and `CGO_ENABLED=0` builds — the + things a Go library is expected to keep — and buys 74 ns against the generated code. + +Four make it well, because they already ship native extensions as a matter of course: Python, +Ruby, C# and Java. There the speedup is 3.9× to 21×, and the pipeline is one their maintainers +have built before. + +### 7. Every handwritten port is already subtly wrong + +Look at the `valid` column. Over the same 1000 inputs, the handwritten arms disagree with the +spec: Python 333, Ruby 329, Go 334, against 337 for everything generated. + +These are not bugs I planted. I wrote each handwritten arm the way the language invites: +`cpf.strip()` in Python strips a different set from JavaScript's `trim()`; Ruby's `String#strip` +handles only ASCII whitespace plus NUL; Go's `strings.TrimSpace` is Unicode space separators, which +excludes `U+FEFF`. Three languages, three different answers, for a CPF a user pasted with an +invisible character in front of it. + +The generated arms all answer 337 because the character set came from the spec, not from the +standard library's idea of whitespace. **The reason to generate is correctness; the performance is +what makes it affordable.** + +--- + +## The tools, surveyed + +| Tool | What it does | Languages | Why it does or does not fit | +| -------------------------------------------------------------------------------- | ------------------------------------------- | ------------------------------------------------------------------------ | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| [UniFFI](https://github.com/mozilla/uniffi-rs) | Rust → bindings, from proc macros | Kotlin, Swift, Python, Ruby; C#/Go third party | **Measured: 3.6× slower than plain Python.** No JS, no Erlang. Built for coarse APIs | +| [Diplomat](https://github.com/rust-diplomat/diplomat) | Rust → FFI, one bridge, many backends | C, C++, JS/TS, Dart, Kotlin, Python | Closest fit technically (ICU4X ships JS via wasm with it), still a binary core with the costs in §5; no Go, Ruby or Erlang | +| [wit-bindgen](https://github.com/bytecodealliance/wit-bindgen) + Component Model | An IDL, guests and hosts generated | Growing; `jco` transpiles components to JS | The most future proof binary story. Still a binary, and the JS output is a wasm blob, not tree shakeable source | +| [Extism](https://extism.org) | Wasm plugin framework, host SDKs everywhere | JS, Go, Ruby, Python, C#, Java, **Erlang/Elixir**, PHP, OCaml, Zig… | Broadest host coverage by far. Adds a plugin protocol on top of the per call cost in §2–3 | +| [wasm2c](https://github.com/WebAssembly/wabt/tree/main/wasm2c) | Wasm → C, compiled natively, no runtime | any language with a C FFI | Removes the _runtime_ dependency, keeps the _binary_ one. Interesting for Erlang NIFs | +| [wasmex](https://github.com/tessi/wasmex) | Wasmtime as an Erlang/Elixir NIF | Erlang, Elixir | The only realistic wasm route for the BEAM | +| [Kaitai Struct](https://kaitai.io) | Declarative spec → parsers in 11 languages | C++, C#, Go, Java, JS, Lua, Nim, Perl, PHP, Python, Ruby | **The precedent for the approach in this repository.** Binary formats only, so not reusable directly, but it proves a spec-to-source compiler scales to a dozen targets | +| [Haxe](https://haxe.org) / [Fable](https://fable.io) | One source language → many targets | JS, Python, C#, Java, PHP, Lua, C++ / JS, TS, Python, Rust, Dart, Erlang | Both emit code that depends on their own runtime library, and neither produces something a Go or Ruby maintainer would review. Fable is worth a footnote because the .NET package is already F# | +| SWIG | C/C++ → bindings | many | Same binary tradeoffs, older ergonomics | + +**Nothing off the shelf does what this project needs**, because the need is unusual: ~100 tiny pure +functions, six ecosystems that each demand idiomatic naming, zero runtime dependencies, and a +flagship package that must stay tree shakeable. Every binding generator optimises for the opposite +shape. The closest thing to prior art is Kaitai Struct, and its model — a declarative spec plus one +compiler backend per language — is exactly what [`spec/bridge`](bridge/README.md) is, at +390–660 lines per target. + +--- + +## Recommendation + +The unit being shared is the **utility's implementation**, not the published package: every +ecosystem writes its own DX by hand — its naming, its option objects, its types, its docs — +over whichever core it gets. That is what makes a hand-written binding acceptable, and it is +why the answer can differ per ecosystem without the library becoming incoherent. + +1. **Write the utility once**, in the portable TypeScript subset + ([`spec/bridge`](bridge/README.md)). One implementation, one review, one set of conformance + vectors. +2. **Emit TypeScript for npm.** Not negotiable: the package is tree shakeable and a binary core + is not. 509 bytes per utility that disappear when unused, against 9,291 indivisible bytes. +3. **Emit Rust for the shared core**, compiled to a C ABI. This is the piece that does not + exist yet: the Rust emitter produces a library crate, and a `cdylib` with + `#[no_mangle] extern "C"` wrappers is a small addition to it. +4. **Bind to that core from Python, Ruby, C#, Java and Erlang.** 21×, 18×, 4.3×, 3.9× and + unmeasured respectively, over a boundary of 1–65 ns. Each binding is hand-written, small, + and lives with the DX layer it serves — the same place those packages already keep their + hand-written code. +5. **Give Go generated source.** cgo's cost is not its 58 ns. It is cross compilation, static + binaries and `CGO_ENABLED=0`, all of which a Go library is expected to keep, traded for + 74 ns against the generated code. The wrong trade. +6. **Keep the other emitters.** They are written and they pass conformance, so they stay as the + escape hatch for any ecosystem that would rather not ship a native artifact — and as the + reference the bindings are checked against. + +The two halves stay honest about each other the same way they do now: every target, generated +or bound, replays the same vectors recorded from the package this repository ships. + +### What the binding route costs to ship + +Worth pricing before agreeing to it, because this is where generated source is free and a +native core is not. + +| Ecosystem | Artifact | Notes | +| --------- | ------------------------------------------- | ------------------------------------------------------------------ | +| Python | wheels per platform × arch, `abi3` | cibuildwheel; ~10 wheels; an sdist fallback needs a compiler | +| Ruby | native gem per platform, or source gem | rake-compiler-dock; a source gem compiling on install is normal | +| C# | NuGet with `runtimes/{rid}/native/` | well-trodden; `SuppressGCTransition` needs .NET 5+ | +| Java | JAR with the library per os-arch, extracted | **needs JDK 22+** for FFM without `--enable-preview`; JNI below it | +| Erlang | a NIF, built on install | unmeasured here; the same C boundary | + +None of that is exotic — every one of those ecosystems ships native extensions routinely — but +it is a release pipeline per package, and it is the reason step 6 exists. + +## What would change the answer + +- A utility whose work is genuinely heavy (a full NF-e XML parse, a large dataset search). Then the + boundary stops dominating. +- A host runtime fixing its call overhead — a wasmtime-py that costs 500 ns instead of 30 000 ns + would make the wasm core competitive per call in Python. +- Go removing the cost of cgo, or the library deciding it does not care about `CGO_ENABLED=0`. + The nanoseconds already favour the binding there; nothing else does. +- The Component Model reaching the point where `jco` emits tree shakeable JS. Today it does not. + +## Reproducing + +The harness that produced every number above — the Rust core, the wasm and `cdylib` builds, the +handwritten and generated arms in six languages, the CPython and Ruby C extensions, the Panama +and P/Invoke arms, the UniFFI crate and the recorded `results.jsonl` — lived at `spec/bench/`. +It was removed once it had answered the question, so that what stays in the tree is the engine +rather than the experiment. It is one command away: + +```bash +git checkout 3780bd6 -- spec/bench # the last commit that carries the harness +bash spec/bench/run-all.sh > spec/bench/results.jsonl +bash spec/bench/run-native.sh >> spec/bench/results.jsonl +python3 spec/bench/table.py # folds the lines into the tables above +``` + +`run-all.sh` also needs `spec/codegen`, the first prototype, which the same commit carries; the +two were retired together, and the paragraph below §1 says what that means for that finding. + +Requires `node`, `python3` (with `wasmtime` and its development headers), `ruby` (with the +`wasmtime` gem and its development headers), `go`, `cargo` with the `wasm32-unknown-unknown` +target, `javac` (21+), `dotnet` (8+) and a C compiler. diff --git a/engine/docs/decisions/0001-authoring-language.md b/engine/docs/decisions/0001-authoring-language.md new file mode 100644 index 000000000..2c4ad100d --- /dev/null +++ b/engine/docs/decisions/0001-authoring-language.md @@ -0,0 +1,48 @@ +# 0001 — The authoring language is a restricted TypeScript with its own checker + +## Context + +The engine needs one source language in which each utility is written once. It has to be a +language contributors already use, it has to support the semantic model (nominal refinements, +integer ranges, effects), and its frontend has to be replaceable later without touching anything +downstream. + +## Options + +**AssemblyScript.** Reuses a ready type checker, but the language cannot express the model: union +types are unsupported by design except `ClassType | null`, so there are no discriminated unions +and no string-literal unions; closures cannot capture locals, so pure lambdas for combinators do +not work; `&&` and `||` always produce `bool`; its integer types are WebAssembly machine types +with wraparound, the opposite of range-proven integers; and its resolver is built for Binaryen +code generation rather than published as a stable library. The part it would save — name +resolution and basic typing of a small subset — is the cheap part; nominal semantic types, ranges, +refinements and effects have to be built either way. + +**Rust as the authoring language.** Better semantics, but extracting types needs +rust-analyzer-grade infrastructure, ownership is noise for pure library logic, and the reference +behavior lives in an npm package. + +**A custom DSL.** Maximum control, but the editor, the language server and highlighting all have +to be built from nothing. + +**Restricted TypeScript with our own semantic checker.** Contributors already read it, `tsc` gives +editor support through a declarations-only prelude, and `oxc-parser` gives a fast, stable syntax +tree. The semantic checker is ours, so the semantics are ours. + +## Decision + +Author in a restricted subset of TypeScript. Parse with `oxc-parser`. Check with our own semantic +checker. `tsc` runs only as an editor and lint aid over the source, backed by +`prelude/index.d.ts`, which declares `Int`, `Digits`, `Ascii` and the intrinsic modules and is +never executed. + +AssemblyScript's "portable code" idea — one source that also type-checks under `tsc` through type +aliases — is adopted in that prelude. + +## Consequences + +- The subset has to be enforced, not assumed: every rejected construct has a diagnostic and a test. +- The Semantic HIR is the frontend contract, and a boundary test proves nothing after it imports + the parser, so a DSL frontend can replace this one later. +- `tsc` sees `Int` as `number`, so it cannot catch a range error; that is the checker's job, and + the checker is the one that decides whether a program compiles. diff --git a/engine/docs/decisions/0002-hand-written-parser-rejected.md b/engine/docs/decisions/0002-hand-written-parser-rejected.md new file mode 100644 index 000000000..99e37587e --- /dev/null +++ b/engine/docs/decisions/0002-hand-written-parser-rejected.md @@ -0,0 +1,28 @@ +# 0002 — `oxc-parser` rather than a hand-written parser + +## Context + +The engine has exactly one external dependency, and it is the parser. A hand-written parser for +the subset would remove it. + +## Options + +**Hand-written parser.** No dependencies at all, and the grammar would be exactly the subset. But +it would diverge from TypeScript in small ways forever, and every divergence is a confusing error +for a contributor who writes ordinary TypeScript. + +**`oxc-parser`.** A real TypeScript parser, fast, with spans on every node and a stable +ESTree-shaped output. It costs one dependency, pinned in `package.json` and +`toolchain.lock.json`. + +## Decision + +Use `oxc-parser`, and keep it behind the frontend boundary so the cost is contained: exactly one +module imports it, and `tests/boundary.spec.ts` proves that. + +## Consequences + +- Source that is valid TypeScript parses like TypeScript, and the subset is enforced semantically + rather than syntactically, which makes the diagnostics better ("`while` is outside the subset; + use a counted `for`") than a parse error would be. +- Replacing the frontend later means replacing one module. diff --git a/engine/docs/decisions/0003-checker-lowers.md b/engine/docs/decisions/0003-checker-lowers.md new file mode 100644 index 000000000..dde98724b --- /dev/null +++ b/engine/docs/decisions/0003-checker-lowers.md @@ -0,0 +1,30 @@ +# 0003 — The semantic checker lowers to Core in the same pass + +## Context + +The plan separates "check the HIR" from "lower the HIR to Core". In practice every typing decision +*is* a lowering decision: which intrinsic `+` resolves to depends on the operand types; whether +`str.codeAt` is allowed depends on a proven length; whether a local needs an unwrap depends on +flow-sensitive narrowing. + +## Options + +**Two passes.** A checker annotates the HIR, then a lowering pass reads the annotations. Every +decision is made once and recorded, and the recording is a second data structure to keep in sync. + +**One pass.** The checker emits Core as it types, so a decision is made exactly where the +information exists. + +## Decision + +The checker types the HIR and emits the annotated Core in one pass (`src/core/check.ts`). The HIR +remains the frontend contract: it is produced by the frontend alone, it carries no target +knowledge, and nothing downstream sees a TypeScript node. + +## Consequences + +- There is no separate "annotated HIR" to keep consistent with the Core. +- Range analysis, refinement narrowing and effect inference are part of checking rather than a + later pass, which is why a refinement can gate an intrinsic signature directly. +- `docs/semantics.md` describes the language; the checker is its only implementation, so the + tests in `tests/refinements.spec.ts` and `tests/subset.spec.ts` are the specification's teeth. diff --git a/engine/docs/decisions/0004-specialization.md b/engine/docs/decisions/0004-specialization.md new file mode 100644 index 000000000..ec4abc306 --- /dev/null +++ b/engine/docs/decisions/0004-specialization.md @@ -0,0 +1,35 @@ +# 0004 — Library helpers are specialized per call site + +## Context + +A refinement is only useful if it survives a function call. `digitAt(value, index)` needs `value` +to be proven long enough for `index`; the caller knows that (its CPF is 11 digits), the helper's +declared signature does not. + +## Options + +**Dependent refinements.** `digitAt(value: Digits, index: Int[0..len(value) - 1])` would express +it exactly, and would turn the checker into a refinement type checker with an SMT-shaped core. + +**Checked accessors everywhere.** `seq.at` answers an `Option`, so nothing needs proving. Correct, +but it pushes a bounds check into every access, including the ones a caller has already proven. + +**Specialization.** Check a helper once per distinct call-site argument type. + +## Decision + +An exported function in a module at the source root is a **utility**: it keeps its declared +signature, which is the published core API. Everything else is **library code** and is checked per +call site, with the caller's proven types substituted for the declared ones (the declared types +still have to accept them). Specializations are capped at 16 per function. + +`seq.at` exists as well, for the genuinely dynamic case: an index derived from a scan. + +## Consequences + +- A helper can serve an 11-digit CPF and a 14-digit CNPJ without either caller losing its proof. +- Several specializations often compile to identical code, so the linker folds functions whose + bodies are structurally identical and joins their parameter types. This is sound because what + the specializations differ in — the refinements — is a proof, not run-time structure, and each + one was proven safe on its own arguments. +- A helper that is never called is never checked, and never generated. diff --git a/engine/docs/decisions/0005-race-is-option-shaped.md b/engine/docs/decisions/0005-race-is-option-shaped.md new file mode 100644 index 000000000..93f42b8d2 --- /dev/null +++ b/engine/docs/decisions/0005-race-is-option-shaped.md @@ -0,0 +1,22 @@ +# 0005 — `race` answers the first `Option`, not the first success + +## Context + +The plan specifies `race` as "the first task to succeed, and when all fail an aggregated domain +error whose components are ordered by task index". Aggregation assumes a task can fail, which in +this language means raising a domain error — and the subset has no `catch`, so a racing author +could not have handled the aggregate anyway. + +## Decision + +Every task answers an `Option`. The race answers the first task that answers `some`, or `none` +when none does. The caller decides what `none` means, and raises its own domain error with its own +message. + +## Consequences + +- `getAddressInfoByCep` raises `GetAddressInfoByCepNotFoundError` itself, which is exactly what + the published package does. +- No aggregate error type is needed, and the component ordering question disappears. +- The determinism rule is unchanged: under the reference model the winner is the smallest virtual + completion time, ties broken by task index. diff --git a/engine/docs/decisions/0006-http-is-an-option.md b/engine/docs/decisions/0006-http-is-an-option.md new file mode 100644 index 000000000..dc5e7eea4 --- /dev/null +++ b/engine/docs/decisions/0006-http-is-an-option.md @@ -0,0 +1,29 @@ +# 0006 — A failed request is absence, not an exception + +## Context + +`http.request` was specified to fail with `HttpError` on a transport error or a timeout. Retry +policy is supposed to be written in source — but retrying means recovering from that failure, and +the subset has no `catch`, on purpose: exceptions are for domain errors that propagate, and bugs +are proven impossible instead of caught. + +## Options + +**Keep the failure and add `catch` for declared error types.** Adds the one construct the subset +exists to avoid, and makes every backend's error plumbing harder (Go would need to translate a +recovered error back into a value). + +**Answer an `Option`.** A transport error or a timeout is absence. A 4xx or 5xx status stays an +ordinary value, because it *is* an answer. + +## Decision + +`http.request(request): Option`, with the effect `Http` and no `Fail`. + +## Consequences + +- Retry is an ordinary counted loop over attempts, and the 250 ms delay between them is + `clock.sleep`, which the reference model advances virtually. +- A target's capability interface answers an optional response: `Promise` + in TypeScript, `Optional[HttpResponse]` in Python, `*HttpResponse` in Go. +- Nothing in the core needs `try`/`catch`, so nothing in the generated code has it either. diff --git a/engine/docs/decisions/0007-shared-target-ast.md b/engine/docs/decisions/0007-shared-target-ast.md new file mode 100644 index 000000000..abe593b10 --- /dev/null +++ b/engine/docs/decisions/0007-shared-target-ast.md @@ -0,0 +1,27 @@ +# 0007 — One Target AST, three printers + +## Context + +The plan calls for "a real AST per target". TypeScript, Python and Go differ in syntax and in a +handful of structural rules (Go has no conditional expression and returns errors as values; Python +has no statement lambdas), but their *structure* — functions, conditionals, loops, calls, records +— is the same. + +## Decision + +One Target AST (`src/backend/tast.ts`) with per-target printers, plus the structural differences +expressed as parameters of the lowering: + +- `statementTernary` hoists a conditional into statements (Go); +- `errorsAsValues` turns a `Fail` effect into a second return value and hoists fallible calls + (Go); +- `asyncColouring` makes a function that reaches `Http` async and awaits its calls (TypeScript); +- `loopCombinators` decides which combinators become loops rather than calls (Go, and `fold` in + Python). + +## Consequences + +- A new target is a capability table, a printer and four flags, not a new tree. +- The risk this accepts is a target whose structure genuinely does not fit — which is exactly what + [targets/rust-sketch.md](../targets/rust-sketch.md) examines before any Rust code is written. +- Nothing in the shared lowering may branch on a target name; the boundary test enforces that. diff --git a/engine/docs/decisions/0008-stdlib-in-the-source-language.md b/engine/docs/decisions/0008-stdlib-in-the-source-language.md new file mode 100644 index 000000000..d6e090245 --- /dev/null +++ b/engine/docs/decisions/0008-stdlib-in-the-source-language.md @@ -0,0 +1,24 @@ +# 0008 — The portable fallbacks live in the engine's own source-language standard library + +## Context + +Fallbacks have to live in the source language, so that a new target is complete as soon as its +core constructs lower. The plan puts them in the *project's* `source/lib/`, which would make every +project that uses the engine reimplement civil dates and scalar comparison. + +## Decision + +The engine ships `stdlib/`, written in the same restricted subset, compiled alongside every +project under the module prefix `std/`. A portable lowering names one of its functions, and +`src/stdlib.ts` lists the ones a lowering may name — the standard library's public surface. + +## Consequences + +- `date.fromYmd` is Howard Hinnant's algorithm written once, and it is the same code in all three + targets. +- `str.compare` on a string that is not proven ASCII is `std/strings::compareScalars` everywhere, + which is why JavaScript's UTF-16 ordering never leaks into a result. +- The standard library is kept through pruning (a lowering may need it even when no project source + names it) and dropped at generation time when the selected lowerings do not use it. +- Nothing in `stdlib/` may import anything: it is compiled in every project, so it stays + self-contained. A test enforces that. diff --git a/engine/docs/decisions/0009-rust-values-are-owned.md b/engine/docs/decisions/0009-rust-values-are-owned.md new file mode 100644 index 000000000..43faac002 --- /dev/null +++ b/engine/docs/decisions/0009-rust-values-are-owned.md @@ -0,0 +1,105 @@ +# 0009 — Every Rust value the core touches is owned, not borrowed + +**Superseded by [0010](0010-rust-parameters-borrow-where-sound.md)**, which lets a function +parameter borrow when a whole-program pre-pass proves it is only ever read. The two obstacles this +decision found — a candidate's `emit` runs before any function scope exists, and `printModule` +sees one module at a time — are both still true and 0010 does not dispute them; what changed is +*when* the decision that needs the whole program gets made. 0010 computes it once, as a pre-pass +over the full `CProgram`, before either of those narrower views exists, and hands the result to +lowering and printing as data instead of asking them to derive it. Record fields and return types +are still unconditionally owned either way — that part of this decision stands. This record is +kept as originally written, including the reasoning below that a later measurement (see 0010 and +`core/bench/README.md`'s Rust section) showed was too conservative for a parameter passed inside a +loop, because *why* the conservative choice was made once is worth keeping next to what changed it. + +## Context + +`docs/targets/rust-sketch.md` planned parameters as `&str`/`&[T]`/`&Record`, with only results +owned, on the reasoning that "every value in the Core is immutable once it escapes, and a mutable +local never escapes, so parameters borrow and results own" — the frozen-on-escape rule doing the +work that would otherwise make borrow inference "the hard part of the backend." + +That reasoning holds for a single function in isolation. It does not survive the shared pipeline +this engine actually has: + +- A candidate's `emit` (the capability table) runs once, during `lowerProgram`, for the whole + program — before `printModule` has run for *any* module. A native or library lowering (`str.trim`, + `seq.at`, …) can itself contain a nested call to an ordinary Core function + (`str.padStart(value, patternSlots(pattern), "0")`, say), and at the point that nested call is + printed, no per-function scope — which locals are already bound, and to what type — exists yet to + consult, because scope is print-time state and this is lowering time. +- `Backend.printModule(module: TModule): string` sees one module at a time, with no access to + sibling modules' function signatures. Deciding "does this argument need `&`" from the *callee's* + declared parameter type — the sketch's own model — needs exactly that cross-module signature + table, which nothing in the shared pipeline hands a backend. +- The Target AST carries no lifetimes at all (`backend/tast.ts` is shared by three targets with no + borrow checker), so there is no way to thread a lifetime parameter through a borrowed signature + even if the first two problems were solved. + +A borrowed-parameter design needs all three of these to hold the way a single hand-written Rust +function would take them for granted. None of them does, in this architecture. + +## Options + +**Borrow parameters, as sketched.** Requires either a whole-program pre-pass (computing every +function's owned-vs-borrowed signature before any lowering runs — a second traversal the other two +targets do not need) or accepting the timing gap above as a known-broken edge case. Neither is a +small addition; the second is a bug, not a design. + +**Own everything, decide ownership only at print time, only where it is unavoidable.** A function +parameter, a struct field, a `Vec` element are all `String`/`Vec`/`T` (owned). Borrowing appears +only where Rust supplies one for free: a method's `&self` receiver, a `for` loop's `.iter()`, a +`std`-provided helper's own `&str`/`&[T]` signature (which this backend's support functions use +throughout, and which candidates borrow into explicitly with `borrowed()`). The one recurring cost +is `.to_owned()` at the handful of positions that actually build an owned value — a `let`, a +`return`, a record field, a list item, `Some(...)`, a call argument — and those are exactly the +positions `docs/semantics.md` already treats as escape points, so the sketch's frozen-on-escape +insight is not wasted, only relocated from "which parameters borrow" to "which expressions need +`toOwned`. + +## Decision + +Every heap-typed value — parameter, local, struct field — is owned. `toOwned(expr, expectedType)` +converts a bare name, a field read or a string literal into an owned value at exactly the positions +that need one; everywhere else, `print` is used directly and never allocates on its own account. +`toOwned` reaches for `.to_owned()` uniformly, never `.clone()`: the blanket `impl +ToOwned for T` makes `.to_owned()` exactly as correct on an already-owned value (a clone) as on a +reference (`&str` → `String`), so the printer never has to know which one it is looking at, and +`clippy::clone_on_copy` — which matches the method name `clone`, not `to_owned` — never fires on a +`Copy` value that happened to go through this path. + +A related simplification travels with this one: `CoreError` is a single flat enum (one variant per +declared domain error), not one error type per utility the way Go's `error` interface and each +target's own error hierarchy might suggest. Every fallible function returning `Result` +is what makes the shared lowerer's Go-shaped `multiLet` + `if err != nil` hoist collapse into a +plain `f(...)?` at print time — there is never a type for `?` to bridge, because there is only ever +one error type in the whole program. + +## Consequences + +- **A few more clones than a hand-tuned Rust port would write.** A value passed to two different + calls is `.to_owned()`'d at both, not moved at the last one; the backend does not attempt a + last-use analysis. Given the actual generated code (nine document-formatting and lookup + utilities, not a hot loop over millions of records), this is the right place to spend simplicity + rather than the wrong one to spend allocations. +- **`opt.unwrap` needs a second form.** `Option::unwrap` takes `self` by value, unlike a Go pointer + dereference, which is free to repeat. A Core-narrowed field read (`if x === undefined { return }` + then `x.field`, possibly several times) re-inserts `opt.unwrap` at every read, since the Core + never re-types the local narrower — so a narrowed read goes through `.as_ref().unwrap()` (a + borrow) instead of a bare `.unwrap()` whenever it is immediately followed by a field access. See + `docs/targets/rust.md`'s note on the same lowering. +- **`task.race`'s closures borrow, not `move`.** A `move` closure would take ownership of whatever + it captures, and two tasks built from the same argument (`fetch_via_cep(cep, env)` and + `fetch_brasil_api(cep, env)`, both closing over `cep`) cannot both move it. Since every capture + here is read-only, an ordinary (non-`move`) closure borrows instead, and any number of tasks can + share the same borrow. +- **Support functions (`support.rs`) still take `&str`/`&[T]`.** They are called the way `std`'s + own functions are — read-only, any number of times, including from inside a loop — which is + exactly the shape a borrow is for. A generated *project* function cannot always make the same + promise (an owned parameter may need to be stored, not just read), which is why the two halves of + this backend's own code borrow and own respectively, on purpose, not by oversight. +- This is the concrete way `docs/targets/rust-sketch.md`'s falsification exercise was wrong, and + the interesting result of carrying it out: not the two frictions it predicted (HTTP, regex — both + confirmed, both solved without a crate), but a third one it did not see, because the sketch + reasoned about one function at a time and the actual friction is a property of the pipeline that + lowers the whole program before any one function's text is final. diff --git a/engine/docs/decisions/0010-rust-parameters-borrow-where-sound.md b/engine/docs/decisions/0010-rust-parameters-borrow-where-sound.md new file mode 100644 index 000000000..fe317cc73 --- /dev/null +++ b/engine/docs/decisions/0010-rust-parameters-borrow-where-sound.md @@ -0,0 +1,170 @@ +# 0010 — Rust parameters borrow where a whole-program pre-pass proves it is sound + +**Supersedes [0009](0009-rust-values-are-owned.md).** + +## Context + +0009 was right about both of the obstacles it named: a candidate's `emit` runs once, during +`lowerProgram`, for the whole program, before `printModule` has run for *any* module, so it cannot +consult a callee's parameter types from a per-function scope that does not exist yet; and +`Backend.printModule(module: TModule): string` sees one module at a time, with no access to +sibling modules' signatures. Neither of those is a mistake in 0009's reasoning, and this decision +does not relitigate them. + +What 0009 missed is that both obstacles are about *when* a decision is made, not about whether the +information exists. `lowerProgram` receives a `CProgram` with every function in it — every +signature, every call, every module — before it lowers a single one of them. The two problems 0009 +found are both instances of one fact: deciding "does this parameter borrow" *during* lowering or +*during* printing does not have enough context. Deciding it *before* lowering starts, over the +whole program at once, does. + +The cost 0009 accepted for owning everything was not hypothetical. `docs/targets/rust.md` and +`core/bench/README.md`'s Rust section measured it directly: `cpf_check_digit`'s loop called +`digit_at(cpf.to_owned(), index)` once per weight (9 to 11 times per call), because `digit_at` +took `value: String` and every call argument was cloned to build it. That loop was 44.9 ms of the +78.0 ms `is_valid_cpf` cost at 200 000 iterations — more than half the call, for a value that +`digit_at` and `cpf_check_digit` both only ever read. `cnpj_check_digit` has no such cost, because +its own source indexes `cnpj.as_bytes()[index]` directly instead of calling a `String`-taking +helper in a loop; that accident of which loop-body shape `core/source` happens to use is the entire +reason `isValidCnpj` (3.59x the handwritten crate) looked so much better than `isValidCpf` +(10.89x) despite validating more digits. Generated Go, on the same benchmark, was already faster +than generated Rust on both rows — a systems language with automatic memory management beating one +with an ownership model it was not using. + +## Decision + +A pre-pass (`analysis/borrows.ts`, `computeBorrowableParams`) walks the whole `CProgram` once, +before any target-specific lowering runs, and decides which `String`- and `Enum`- and `List`-typed +function parameters may be printed as `&str`/`&[T]` instead of `String`/`Vec`. The search starts +**optimistic**: every eligible parameter is assumed borrowable, and a parameter is **demoted** to +owned the moment direct evidence shows it needs to be — returned, stored into a record field, a +list element or a thrown error's payload, assigned to, or forwarded unchanged to a parameter of +another function that already needs to be owned. That last rule is a fixpoint over the call graph: +whenever a callee's parameter is found to need ownership, every caller that forwards its own +parameter into it, bare and unchanged, is demoted too, and the demotion is applied with a worklist +over the reverse call graph until nothing more changes. Starting optimistic and demoting is sound +here specifically because demotion only ever *adds* to the owned set — never removes from it — so +the set of "still borrowable" parameters shrinks monotonically to a fixpoint no matter what order +the evidence is found in; `docs/semantics.md` §7 forbids recursion, so the call graph has no +cycles and the worklist provably terminates (each key enters the queue at most once). + +The result — a `BorrowMap`, from a function's fully qualified name to the names of its own +borrowable parameters — is handed to the shared lowerer as an input (`LowerOptions.borrows`) and +to the Rust printer as data already baked into the Target AST, not as something either of them +computes: + +- `Lowerer.lowerFunction` looks the map up once per function, before lowering that function's body, + and sets `TParam.borrowed` on each parameter accordingly. `printFunction` reads that field to + decide `&str`/`&[T]` vs. `String`/`Vec` in the signature; it does not re-derive it. +- Every "local" reference the lowerer builds (`Lowerer.expr`'s `"local"` case) carries `borrowed: + true` when it names one of the *current* function's own borrowed parameters. This is the fix for + 0009's exact timing problem: a candidate's `emit` can now ask a plain data field on the `TExpr` + it already has in hand, at whatever point during lowering it runs, instead of needing a + per-function scope that does not exist yet. +- Every "call" node carries `borrowedArgs`, one flag per argument, set from the *callee's* entry in + the same `BorrowMap` at the point the call is lowered (`Lowerer.expr`'s `"call"` case already has + the callee's `CFunc` in hand there). The printer's `call` case reads it directly: an argument the + callee only reads is passed with `&`, everything else still goes through the existing + `toOwned`/`callArg` path, unchanged. +- The Rust backend supplies the pre-pass to the shared pipeline through one new, optional hook on + `Backend` (`extraLowerOptions: (program) => Partial`), merged into `LowerOptions` + before `lowerProgram` runs. `generate.ts` stays target-independent — it calls the hook if the + backend defines one and does nothing otherwise — and Go, Python and TypeScript, which define no + such hook, are unaffected in every sense that matters: same code path, same output, same + conformance, because `LowerOptions.borrows` is `undefined` for them and every place that reads it + treats `undefined` as "owned", exactly 0009's model. + +Record fields and every function's return type stay owned, unconditionally — this decision does +not touch either. A borrowed field would need a lifetime parameter on the struct itself +(`struct Foo<'a> { name: &'a str }`), which is a materially larger change (every consumer of that +struct now carries a lifetime too) for a benefit this task was not measuring; a borrowed return +type is not expressible without one either, since nothing borrowed can outlive the call that +produced it. `toOwned` already converts a borrow to an owned value uniformly at both positions +(`&str`'s `.to_owned()` and `String`'s `.to_owned()` are the same call, printed the same way), so +nothing about them needed to change for parameters to start borrowing. + +### Why this is sound, position by position + +Every position that builds an owned value in the Rust backend — a `let`, a `return`, a record +field, a list item, `Some(...)`, an owning call argument — already goes through `toOwned`, and +`toOwned`'s `.to_owned()` on a name is exactly as correct whether that name is a `String` or a +`&str` (0009's own point about `.to_owned()` vs. `.clone()`, unchanged by this decision). That is +what makes the four demotion rules a genuine *optimization* question rather than a *correctness* +one: nothing here is unsound to skip, in the sense of producing code that fails to compile. What +each rule actually buys is not needing a clone at all where a bare move or a bare borrow now +suffices instead: + +- **Returned, or reaches a return through another value.** Returning an owned parameter can be a + move (`return param;`, free); returning a borrowed one needs a clone (`return param.to_owned();`) + because nothing borrowed can be handed out past the call. Keeping it owned keeps that path free. + A local that is a bare, untransformed alias of the parameter (`let x = param;`) inherits the same + treatment, because at the exact point that alias was built, it went through `toOwned` already — + whatever the parameter's own status, that clone happened once, there; nothing about the + parameter's declared type changes after that. +- **Stored into a record field, a list element or a thrown error's payload.** Same shape as above: + a field, an item or an error payload is always built owned, so the parameter being owned instead + of borrowed only matters for whether that specific write can be a move. +- **Assigned to.** `check.ts` marks every parameter binding `mutable: false`, so this never actually + fires against an unshadowed parameter today — a mutable local can share a parameter's name only + by shadowing it with its own `let`, which the alias tracking in `analysis/borrows.ts` already + treats as a break in the alias (a fresh, independent, owned local from that `let` on). The rule + is kept anyway, as a static safety net: if the source subset ever admits a form of parameter + mutation, this rule already demotes the right parameter without anyone having to remember to add + it then. +- **Forwarded to a parameter of another function that is itself owned.** A thin forwarding function + (`f(s) { return g(s); }`, `s` used nowhere else) that keeps its own parameter owned can hand `s` + to `g` as a move if `g` needs it owned, instead of paying a clone at that internal call; keeping + `f`'s own parameter borrowed would force that clone every time `f` is called instead. This is + the one rule that is not purely local to one function, which is why it is a fixpoint over the + call graph rather than a single pass over one function's body. + +Two things this pass deliberately does not chase, both because they cost nothing to leave alone, +given the design above: + +- **Last-use elision.** `toOwned`'s `"name"` case still clones unconditionally on every occurrence, + even the last one, even for an already-owned local — the same simplification 0009 named and + accepted ("the backend does not attempt a last-use analysis"). This decision does not add one. + What it removes is the clone an *owned call argument* needed only because the parameter itself + was declared owned when the callee never asked for that; it does not remove the clone a + genuinely-owned value pays when it is used more than once. +- **An intrinsic's own operands.** `str.concat`, `str.padStart` and the rest of `RUST_CANDIDATES` + already borrow their own arguments, through the existing `borrowed()` helper, independently of + this map — that is the "Support functions still take `&str`/`&[T]`" half of 0009, untouched. + What *did* need a fix, because parameters can now genuinely be `&str`/`&[T]` themselves, is + `borrowed()`'s own `"name"` case: wrapping an already-borrowed name in another `&` would print + `&&str`/`&&[T]` — harmless at compile time (Rust's coercion resolves it) but exactly what + `clippy::needless_borrow` exists to flag under `-D warnings`. `borrowed()` now checks the same + `borrowed` field `Lowerer.expr` sets on every "local" reference (and the existing + `hoistedConstantNames` check, for a hoisted `&'static [T]` table) before deciding to add a `&`. + +## Consequences + +- **`is_valid_cpf`'s loop cost is gone, not reduced.** `cpf_check_digit`'s own parameter, and + `digit_at`'s, are both borrowable (each only reads its `value`), so `cpf_check_digit`'s loop is + `digit_at(cpf, index)` — a bare `&str` copy, not a clone — at every iteration. Re-measured, + `cpf_check_digit(_, 9)` and `cpf_check_digit(_, 10)` together cost about 2.2 ms over 200 000 + calls, against 44.9 ms for one of the two before this decision. +- **The new bottleneck is `keep_digits`, not ownership.** With the loop's clones gone, + `is_valid_cpf`'s single remaining allocation-heavy step is `keep_digits`'s own + `.chars().filter().collect::()`, building the fresh, necessarily-owned value the function + returns — about 9-9.5 ms of `is_valid_cpf`'s ~16.8 ms, next to `re.test`'s ~4.4 ms (unchanged; + this decision does not touch regex) and well under 1 ms for `is_repeated` plus the now-free + `cpf_check_digit` calls. `core/bench/README.md`'s Rust section has the full attribution and the + before/after benchmark. +- **Generated Rust now beats generated Go on both CPF and CNPJ rows** (it did not, before this + decision), the specific regression this decision exists to close. +- **A hand-written port would still borrow a few things this pass keeps owned.** The alias tracking + in `analysis/borrows.ts` only follows a *bare* `let x = param;` as a continued alias; a parameter + threaded through a `Record` field, or read inside a combinator lambda in a way more indirect than + a bare capture-and-return, is not traced past that point, and the record/list/error-payload rule + above is evidence-based rather than data-flow-complete (it looks at the immediate write, not at + what later happens to the container). Where this pass cannot prove a parameter's status either + way through those paths, it does not demote it — the parameter simply stays eligible and ends up + borrowed if nothing else demotes it first, or a hand-written port would sometimes make a sharper + call than this pass does in the other direction, on a value it can see is never touched again + after a store this pass conservatively still credits toward "escapes". Both directions are safe; + neither is exploited by anything measured in `core/source` today. +- **Go, Python and TypeScript are untouched.** `LowerOptions.borrows` and the `borrowed`/ + `borrowedArgs` Target AST fields are optional and target-agnostic by construction; none of the + other three backends reads them, none of their `extraLowerOptions` hooks exist, and their + generated output and conformance are unchanged by this decision. diff --git a/engine/docs/decisions/0011-public-entry-points-vs-capabilities.md b/engine/docs/decisions/0011-public-entry-points-vs-capabilities.md new file mode 100644 index 000000000..09219fc76 --- /dev/null +++ b/engine/docs/decisions/0011-public-entry-points-vs-capabilities.md @@ -0,0 +1,126 @@ +# 0011 — A public entry point never takes capabilities; a named seam does + +## Context + +Capability threading (`analysis/capabilities.ts`) infers which functions reach `Http`, `Clock` or +`Random` and adds an `env` parameter to exactly those, bottom-up over the call graph. That pass +was correct on its own terms — it is what lets `core/source/get-address-info-by-cep.ts` call +`http.request(…)` without ever naming an environment — but nothing stopped its output from +reaching the printed signature of an exported utility. `getAddressInfoByCep(cep: string)` in +`core/source` generated `getAddressInfoByCep(cep: string, env: Capabilities)` in every target; +`generateCpf(): string` generated `generateCpf(env: Capabilities): string`. A project treating +`core/out/` as a drop-in replacement for the package `core/source` re-implements cannot +call it the way it calls the original — every capability-taking utility needs code changed at the +call site, for a parameter the source never declared and a caller has no way to construct +correctly (get the fixture-shaped fake instead of a real environment, and every value returned is +garbage or a transport error). + +Two shapes were available once the defect was named: + +1. **Keep threading, but stop it from reaching a utility's own signature** — thread `env` between + internal helpers exactly as today, and give every utility a wrapper that supplies the + capability itself, under a name that is not the utility's own. +2. **Stop threading at the top: make every utility a "root" the capability graph cannot cross**, + forcing each one to build its own environment inline, at the top of its own body, with no + internal function ever receiving one as a parameter. + +Shape 2 was rejected. It would still need something to build an environment from nothing before +any of the utility's own logic runs — the exact problem this decision exists to solve, just moved +one call frame earlier — and it would also have to rebuild that environment on *every call*, since +a function with no capability parameter has nowhere to receive a shared one from; the resulting +generated code would call `defaultCapabilities()` (or its `Capabilities()`/`newEnvironment()` +equivalent) once per invocation, which is both a real per-call cost (a fresh `AbortController`, a +fresh dict, in a case where nothing about the environment changed) and a availability +regression against the actual published packages: they draw randomness through a bare +`Math.random()`, no allocation and no construction step per call, so their published cost floor is +lower than any shape that reconstructs an environment on every entry. + +Shape 1 was chosen. + +## Decision + +An exported utility whose effects reach `Http`, `Clock` or `Random` is split, after lowering, into +two Target AST functions in the same generated module (`backend/generate.ts`, +`splitCapabilityEntryPoints`, run once per target's whole output): + +- **The public wrapper** keeps the utility's own name and exactly the parameters and return type + its source declares — no capability parameter, ever. Its body is one call: the seam below, with + the source's own arguments forwarded unchanged plus one more, a reference to a **module-level + singleton**, built once when the module loads, never reconstructed per call. `defaultCapabilities()` + itself (TypeScript) or `Capabilities()` (Python) is still generated, exactly as before, precisely + so that singleton has something to be built from once; what changed is that nothing calls it more + than that one time. +- **The seam** is the function capability threading actually produced: the original name with a + suffix that reads as what it is in each target's own casing (`generateCpfWith` in TypeScript, + `generate_cpf_with` in Python) — never a generic word like "impl" or "internal" that carries no + information about which capability-taking function this is. It keeps the capability parameter, + keeps its own module's visibility (`export`/no leading underscore) because two things outside its + own module have to reach it — the wrapper is a code-generation detail, not the reason it stays + visible — and is what the differential conformance driver's dispatch table names directly, so + fake capabilities from `fixtures.json` still reach every capability-taking utility exactly the + way they did before this split. `rust: 4256/4256` and the other three targets' conformance counts + are unaffected by this decision for exactly that reason: the driver was never calling the public + surface to begin with, on any target, before or after. + +`API.json` reflects the split directly: `functions` lists the wrapper — the source's own signature, +with no capability parameter and no `"env"` effect — and a new top-level array, `seams`, lists the +capability-taking form by its own name, with `publicName` pointing back at the wrapper and +`hasWrapper: true`. A DX author reading `API.json` sees the utility's real signature where they +would look for it, and sees separately, not folded silently into the same list, that this +particular utility has an internal form taking capabilities, without mistaking that form itself +for a second utility. + +**Go and Rust get no wrapper.** Both generate the `Capabilities` interface/trait only — no HTTP +client, no clock, no RNG, anywhere in the generated *library* (`support.go`, `support.rs`); the +fake either differential driver builds lives in the driver binary (`cmd/driver`, `src/bin/driver`), +never in the crate or package a consumer would import. There is nothing standing in for +`defaultCapabilities()` to build a singleton from, and one is not fabricated for the sake of +producing a wrapper. `splitCapabilityEntryPoints` is target-generic; whether a target gets a +wrapper is entirely decided by whether `Backend.defaultCapabilities` is present, and Go's and +Rust's backend objects simply do not define it. For these two targets, the function +`splitCapabilityEntryPoints` would otherwise have renamed is instead only *marked* (`seam: true`, +`hasWrapper: false` in `API.json`), under its own original name, still taking `Capabilities` +directly as its only form. This is reported as a finding, in `docs/semantics.md` §4.1 and in +`docs/targets/go.md` and `docs/targets/rust.md`, not hidden behind a fake that would make the +generated library depend on something the published Go and Rust ports do not (an HTTP client, a +system RNG) or behave differently from a caller's own environment in ways nothing would flag. + +Why the seam is a real, separate, generated function rather than an inline "if no capability was +given, build one" branch folded into the same signature (the shape a hand-written wrapper might +take, e.g. `getAddressInfoByCep(cep, env = null)`, defaulting internally): the source language this +engine compiles admits no optional parameters and no runtime type branching on "was an argument +supplied" — see `docs/semantics.md` §7's subset rule — so a single function that behaves two ways +depending on whether a caller passed a fourth thing is not a shape any Core function can express at +all, in any target. Two functions, one calling the other, is. + +## Consequences + +- **Every exported utility's signature now matches its own source declaration, name for name and + parameter for parameter**, in every target that can build a default (TypeScript, Python) and, + trivially, in every target where no utility needs one (all four, for the six utilities with no + effects beyond `Fail`). `getAddressInfoByCep(cep)`, `generateCpf()`, `generateCnpj()` now compile + and run as drop-in replacements for `src/get-address-info-by-cep`, `src/generate-cpf`, + `src/generate-cnpj` with no call-site change. +- **A module-level singleton, not a per-call construction.** `core/out/typescript/capabilities.ts` + gains `export const DEFAULT_CAPABILITIES: Capabilities = defaultCapabilities();`; + `core/out/python/_support.py` gains `DEFAULT_CAPABILITIES = Capabilities()` at module scope. Both + are built once, at import time, and every wrapper generated into that target shares the one + instance — the same pattern `core/bench/typescript.ts` and `core/bench/rust/src/main.rs` already + used by hand ("built once, like a real caller would, then reused") before this decision made it + the generated default's own behavior, not something only a benchmark bothered to do. +- **Go and Rust keep a permanent, structural gap from full parity** on exactly the three utilities + whose effects reach `Http`, `Clock` or `Random` (`getAddressInfoByCep`, `generateCpf`, + `generateCnpj`). Closing it for real — not by fabricating a fake — would mean choosing and + vendoring an HTTP client and a randomness source into the generated library for those two targets + specifically, which is a materially larger decision (a new dependency the published Go and Rust + ports do not have to carry, since `net/http` and `math/rand`/`crypto/rand` are already stdlib for + them — so the actual obstacle is a design choice to keep `core`/`coreout` dependency-free rather + than a technical one) than this task's scope, and is named here as the next thing to decide, not + solved by this decision. +- **`core/bench/typescript.ts` and `core/bench/python.py` call the generated `generateCpf`/ + `generateCnpj` with no arguments now**, matching the handwritten side they are timed against + exactly; the explicit `defaultCapabilities()`/`Capabilities()` construction those two harnesses + used to need is gone from them, because the generated wrapper does it internally. Go's and Rust's + bench harnesses are unaffected — both already built their own fake `Capabilities` value by hand, + because neither language's generated library could give them a real one, before or after this + decision. diff --git a/engine/docs/decisions/0012-generated-source-not-a-bound-binary.md b/engine/docs/decisions/0012-generated-source-not-a-bound-binary.md new file mode 100644 index 000000000..7ab34d7b6 --- /dev/null +++ b/engine/docs/decisions/0012-generated-source-not-a-bound-binary.md @@ -0,0 +1,61 @@ +# 0012 — Generated source, not one binary core with bindings + +## Context + +The engine generates native source for every target. The obvious alternative was never written +down: build the logic once — in Rust, in C, in WebAssembly — and give every ecosystem a **binding** +to that one artifact. It is a smaller tool, it has one implementation to review instead of four +emitters, and the binding is code each ecosystem already writes by hand. + +A separate branch measured it rather than argued it, and the measurements are kept in +[`../bindings-or-generated-source.md`](../bindings-or-generated-source.md). Two of its findings +decide this, and one of them decides it unconditionally. + +## The unconditional one + +**A tree-shakeable package cannot take a binary core.** The npm package drops a utility nobody +imports; a WebAssembly module is indivisible and asynchronous to start. Measured on that branch: +one utility as generated source was 509 bytes that disappear when unused, against 9,291 bytes that +cannot be split and cannot be removed. That is not a preference between two acceptable options, it +is a requirement the package already has, so JavaScript gets generated source whatever is true +elsewhere. + +Go is the second case where the currency is not nanoseconds: cgo costs cross compilation, static +binaries and `CGO_ENABLED=0` builds, all of which a Go library is expected to keep, in exchange +for tens of nanoseconds. + +## The one that does not decide it, and is worth knowing anyway + +For Python, Ruby, C# and Java, a **binding is genuinely cheaper than generated source** — a CPython +extension module at 27 ns per call against `ctypes` at 433 ns, P/Invoke at about a nanosecond — and +the first version of that document got this wrong by measuring the bindings a *script* reaches for +rather than the ones a *package* ships. It was corrected on the branch, and the correction is the +reason this decision is scoped rather than absolute. + +So the honest position is not "generated source wins". It is: + +- JavaScript and Go must have generated source, on grounds that are not about speed; +- the other ecosystems could be served either way, and this engine chooses generated source for a + different reason — the output has no runtime, reads like code a person in that language would + have written, and can be read in review without running anything. + +## Decision + +Generate source for every target. Do not ship a binary core. + +Where an ecosystem would rather bind to one, that remains open and this engine does not block it: +its Rust target is `std`-only and carries no runtime, so a `cdylib` with `extern "C"` wrappers over +the generated crate is an addition to the backend rather than a second compiler. Nothing in the +Core, the checker or the other three targets would move. + +## Consequences + +- Four emitters to maintain instead of one core plus bindings, and per-target semantic knowledge + (`str.compare`'s ordering, `charCodeAt`'s UTF-16, `regexp`'s lack of a compilation cache) has to + be encoded four times rather than once. That cost is real and the lowering tables are where it + lands. +- In exchange, the output is readable, tree-shakeable, dependency-free and per-target idiomatic, + and the differential harness can compare four independent implementations against a reference + rather than one implementation against itself. +- The numbers behind this cannot be re-run from this repository: the harness was retired on the + branch that produced them. Anything load-bearing should reproduce them first. diff --git a/engine/docs/decisions/0013-inlining-pays-for-itself.md b/engine/docs/decisions/0013-inlining-pays-for-itself.md new file mode 100644 index 000000000..9eb25ce22 --- /dev/null +++ b/engine/docs/decisions/0013-inlining-pays-for-itself.md @@ -0,0 +1,78 @@ +# 0013 — Inlining has to pay for itself, in the currency the target is judged in + +## Context + +`optimize/inline.ts` runs per target, with a budget each backend sets, because the cost of a call +is not the same everywhere: CPython pays a full frame, V8 usually elides the call once a site is +hot, rustc inlines across crates only under LTO. That part was already measured and is not in +question here. + +What the budget did not have was a price. It said how big a callee could be — "at most N +statements" — and nothing about how many copies of it the program ended up holding. For a target +that is compiled ahead of time, that is fine. For TypeScript it is not, and +[ADR 0012](0012-generated-source-not-a-bound-binary.md) already says why: the npm package this +engine generates for is tree-shakeable, and a consumer who imports `isValidCpf` must not carry +`getHolidays`. Bytes over the wire are a result that package is judged on, not a nicety. + +Measured, with a single-import entry point per utility, bundled and minified by esbuild and +gzipped — what a consumer's bundler would actually produce. The budget of the day +(`maxStatements: 6`, nothing else) against the same program with the pass disabled: + +| export | inlining off | `maxStatements: 6` | +| --- | ---: | ---: | +| `generateCnpj` | 642 | 1,591 (+148%) | +| `generateCpf` | 623 | 1,343 (+116%) | +| `getHolidays` | 872 | 1,403 (+61%) | +| `isBusinessDay` | 1,065 | 1,638 (+54%) | +| `isValidCpf` | 342 | 440 (+29%) | +| every export | 3,224 | 5,644 (+75%) | + +What it bought was two rows of `core/bench`, `generateCpf` and `generateCnpj`, by around a tenth +each — inside the run-to-run noise of a generator whose own retry loop is random. `generateCpf` +was 45 lines of generated TypeScript before the pass and 408 after: nine unrolled copies of a +rejection-sampling loop, one per digit. Neither the speed nor the readability was worth the bytes. + +## Decision + +The budget states what an inline may **add**, and the pass refuses the ones that cost more than +that: + +- **`maxDuplicatedNodes`** caps, per inline, the Core nodes it adds. A target compiled ahead of + time leaves it unset; TypeScript sets it, and what it says is "duplicate only what is small". +- Nodes, not statements, because a one-expression helper is one statement whether it reads `a + b` + or spans half a screen. +- An inline that takes a callee's **last** call site is refunded that callee's whole definition: + nothing reaches it any more, so `backend/lower.ts`'s `closure` drops it and the code moved + rather than multiplied. A sole-call-site helper is therefore absorbed whatever its size, and + `maxDuplicatedNodes: 0` means "take the inlines that pay for themselves", not "inline nothing". +- The cap is per inline rather than a pool for the whole pass, so the answer does not depend on + which call site the walk reached first. + +Two things had to change before that budget could be honest about its own price, and both are +improvements on their own: + +- A callee that is a single `return ` is **substituted as an expression**. Routing it through + the general path bound each parameter to a `let`, opened a synthetic `Option`, assigned through + it and unwrapped it — four statements and a sentinel where the source had an expression, bigger + than the call it replaced. `digitAt(cpf, index)` now becomes `cpf.charCodeAt(index) - 48`. +- A callee whose only `return` is its last statement is spliced **without the sentinel**, and a + parameter passed a literal or a local is **substituted rather than bound**. There is nothing for + a flag to answer in a straight line, and `const _inl117_year = year;` is scaffolding no author + would leave in. + +`engine/scripts/size.ts` measures the result, `verify`'s `typescript size` step fails on a +regression past the budget in `core/out/typescript/SIZE.json`, and `core/bench` measures what the +inlining bought. + +## Consequences + +- TypeScript settled at `maxStatements: 8, rounds: 4, maxDuplicatedNodes: 6`, chosen by sweeping + both against the measurement rather than by argument. Every export is now **smaller than with + the pass disabled** — 3,175 bytes gzipped across all nine against 3,224 — so inlining stopped + being a trade for this target and became a win on both axes. +- Python keeps an uncapped budget (`maxStatements: 12`), which is the same decision reached the + other way: its cost is interpreter frames, its output is not downloaded, and a sweep of + `maxDuplicatedNodes` at 6 and 24 made both of its generator rows slower. Go and Rust still run + no inlining at all, because their compilers do it better. +- The generated TypeScript reads like its source again. That was not the goal, and it is the part + worth keeping: `generateCpfWith` is nine lines, the same nine the author wrote. diff --git a/engine/docs/fuzzing.md b/engine/docs/fuzzing.md new file mode 100644 index 000000000..a1a255066 --- /dev/null +++ b/engine/docs/fuzzing.md @@ -0,0 +1,280 @@ +# Random program generation + +The 64 hand-written cases and the two translation-validation programs are a fixed target. The two +soundness bugs found in this codebase before this generator existed — `break`/`continue` not +reaching a loop's fixpoint, and a switch-case `break` lowered as a loop break — were both found by +a person reading code, not by either of those. This is the search that replaces "read more code." + +`src/fuzz/` generates well-typed programs in the engine's subset and checks them two ways; `scripts/fuzz.ts` +is its command line. + +## Where it lives, and why it has no dependency + +The generator is `engine/src/fuzz/`, inside the engine, not a separate package and not built on +`fast-check` (already a devDependency of the repository root, and the obvious tool for this if the +generator lived at the repository root instead). Two reasons pulled the other way: + +1. **The engine has exactly one runtime dependency (`oxc-parser`), and that restraint is + deliberate** — see the one-line "Deliberately absent" callouts throughout `semantics.md`. Taking + `fast-check` as a second one to generate test input is a strange trade for a project whose whole + design argument is "prove it, don't depend on a library proving it for you." +2. **A property-testing library's arbitraries compose *values*.** What this generator needs to + compose is *statements that stay well-typed as they accumulate a range* — an `Int.map`/`chain` + pipeline does not know that pushing a fourth statement into a loop body has to keep the whole + body provable by the checker afterward. That bookkeeping (`src/fuzz/generate.ts`'s `Ctx`, tracking + which locals are safe to index and which are in scope at all) is the actual content of a program + generator for this language, and `fast-check` would not remove any of it — only its low-level + `Random`/shrinking plumbing would be reused, and that plumbing is a few dozen lines to hand-roll + (`src/fuzz/rng.ts`) against the wall of API surface a real dependency would carry. + +So: a 32-bit seeded PRNG (`rng.ts`, mulberry32 + a splitmix32 sub-seed derivation), a small +generator-owned AST with its own printer (`ast.ts`), a recursive generator over that AST +(`generate.ts`), and a delta-debugging shrinker over the same AST (`shrink.ts`) — a few hundred +lines total, no new dependency, all reviewable the same way the rest of the compiler is. + +## What it generates, and what it deliberately does not + +Every generated program is one exported function, biased toward the shapes that broke the checker +before (see `generate.ts`'s module doc for the exact weighting): + +- loops with `break`/`continue` in every position, including a `break`-then-fall-through-to-`else` + shape that specifically exercises what a loop's fixpoint carries out of a conditional branch; +- an accumulator (`Int`, `boolean`, or a code-point list built into a string) carried across + iterations, including through `break`/`continue`; +- nested loops, up to two deep; +- a `switch` over a generated `Enum`, sometimes nested inside a loop; +- an index into a list, generated *only* where the generator itself can see it is safe — a counted + loop's own counter, ranging over a same-length list's `.length` — which is exactly the shape the + `break`/`continue` fixpoint bug lived in (`docs/semantics.md` section 7's `E_SIGNATURE` example, + and `tests/loops.spec.ts`); +- integer division and modulo with a proven non-zero literal divisor, which is what exercises the + Python backend's truncated-vs-floored divergence (`docs/semantics.md` section 2.1); +- occasionally (one list per function, capped — see `generate.ts`) a list long enough to push a + loop past the 64-iteration exact-fixpoint threshold into the widen/clamp path + (`docs/semantics.md` section 2.2), not just the small exact case the historical bug happened to + trip on; +- an early `return` from inside a loop. + +Every mutable integer local is declared `Int` — the full platform-safe domain — never a tight +`IntRange`. This is not a simplification that dodges the interesting bugs: the checker still infers +and proves a *tight* range for the local on every assignment regardless of its declared ceiling, +and that tight range is exactly what Layer 1 (below) checks the interpreter's real answer against. +Declaring `Int` only means the compiler never rejects a program for accumulating past its author's +stated intent — which is not the analysis this generator exercises — and it is what makes free-form +generation of arithmetic tractable without re-implementing the checker's own interval arithmetic to +predict, before generating an expression, exactly how wide an annotation it would need. + +**Deliberately absent**, to keep the generator's own surface reviewable: records, `Decimal`, +`CivilDate`/`Instant`/`Duration`, `Http`/`Clock`/`Random` capabilities, `task.race`, `throw`, +template literals, and `String`/`Ascii` function parameters (string values only appear as the +*output* of building a code-point list with `str.fromCodePoints`, not as generated input). None of +these are where the known bugs lived, and the bias list above is what the task asked this generator +to hunt for. Extending the generator to any of them is straightforward: add a case to the `Shape` +union and a branch to the relevant `gen*` function in `generate.ts`. + +## The two comparisons + +`src/fuzz/harness.ts` runs each generated program two ways: + +- **`runFast` — checker against reality (Layer 1).** Compiles the program, runs the reference + interpreter on random inputs (`values.ts`'s `randomValue`, biased toward each type's own edges), + and checks that every value it actually produced lies inside the range, length and character + class the checker proved for it (`values.ts`'s `withinType`). No code generation at all, which is + what makes it cheap enough to run in the thousands on every `verify` — see "Wiring" below. This + is exactly the shape of the `break`/`continue` bug: the checker proved a range, and the question + is whether reality (the interpreter, which follows real control flow) agrees. +- **`runFull` — interpreter against all four targets, in both idiom modes (Layers 2/3).** Generates + TypeScript, Python, Go and Rust from the same compiled program and compares every answer against + the interpreter's, reusing `src/conformance/differential.ts`'s `runInterpreter`/`runTarget`/ + `compare` exactly as `core/conformance/run.ts` drives them by hand — this generator does not + reimplement that machinery, only supplies it with generated cases instead of hand-written ones. + Compiling and running four toolchains is slow, so `runFull` first compiles every candidate program + on its own (an inexpensive pass — the same one `runFast` does) and only pays for code generation + and four builds on the survivors, batched into one project so the cost is paid once per batch, not + once per program. + +A program that fails to compile is not a finding: the generator is not required to produce only +well-typed programs by construction (see `generate.ts`'s module doc on why that would mean +re-implementing the checker's own type inference to generate goal-directed expressions), only to +generate *mostly* well-typed ones and let the compiler be the filter. Both modes report the +compile success rate for this reason. + +## Seed and shrink + +Every program is addressed by a `(seed, index)` pair (`rng.ts`'s `subSeed`), so replaying +`--seed --count 1` after adding `--seed` reproduces bit-for-bit the same program the batch +produced at index 0, independent of what batch size or machine produced the original finding. A +report always prints the seed next to the finding for exactly this reason. + +A finding is shrunk before it is reported (`shrink.ts`): + +- **`shrinkInputs`** narrows the failing argument tuple toward the edge of each parameter's own + type (zero, `false`, an empty or shorter list) without touching the program at all — cheap, since + it costs one interpreter call per trial, no recompilation. +- **`shrinkProgram`** narrows the program itself, one delta-debugging pass at a time: drop a + statement, drop an `if`'s `else`, drop a `switch` case. Every candidate is recompiled and rejected + outright if it no longer type-checks, so a shrunk report is always itself a valid program in the + subset. `runFast`'s violations are shrunk this way (`shrinkViolation`); `runFull`'s four-target + divergences are reported with the program the generator produced, unshrunk — recompiling and + rebuilding four toolchains per shrink trial is too slow to run automatically, so a full-mode + finding is a starting point for a person's own minimization, not a final answer the way a fast-mode + one is. + +## Wiring + +`scripts/verify.ts` runs `fuzz fast` with a small, fixed seed on every verification — no code +generation, so it costs seconds, not minutes. `fuzz full` is not wired into `verify`: it is run +deliberately, with a larger budget, exactly like `core/conformance/run.ts` is run deliberately +rather than on every keystroke. + +```sh +node scripts/fuzz.ts fast --seed 20260921 --count 5000 +node scripts/fuzz.ts full --seed 20260921 --count 300 --targets typescript,python,go,rust +``` + +A new target's own conformance story (`adding-a-target.md` step 5) should include a `fuzz full` run +restricted to it (`--targets `) alongside the differential conformance harness: the generator +does not know which target it is comparing, so a new backend gets the same generated coverage the +first four did for free. + +## Findings + +The generator's first runs at scale found four bugs, none of which the 64 hand-written cases or +the two translation-validation programs reached. Three were small enough to fix on the spot and +are already fixed in this tree; the fourth is reported and left, per this project's own rule that a +found bug is the deliverable, not a failure to be smoothed over. + +### Fixed: a `switch`'s effect on a mutable local never left the `switch` (Layer 1) + +`seed 12345 --count 300`, program `program::fuzz_185` (auto-shrunk); minimal form: + +```ts +export type Kind = "A" | "B"; + +export function f(p1: List, 7, 7>, kind: Kind): Int { + let acc: Int = -8; + for (const e of p1) { + switch (kind) { + case "B": + acc *= -2; + break; + default: + break; + } + } + return acc; +} +``` + +`core/check.ts`'s `switchStatement` collected the type-state at the end of every case (`exits`) — +used to check exhaustiveness — but the state was never merged back into the enclosing scope: the +function restored the pre-switch snapshot and returned without ever calling `applyJoin`. The +checker proved `f`'s return was `Int[-8..-8]` (the switch, in its model, changes nothing) while the +interpreter, calling `f([0,0,0,0,0,0,0], "B")`, actually returned `1024` — `-8 * (-2)^7`, seven real +multiplications the checker never saw. This is the same shape as the historical `break`/`continue` +bug the checker's tests already guard against — a control-flow construct's effect not reaching the +point after it — just in `switch` instead of a loop. Every `if`/`else` in the same function was +already threaded correctly; `switch` was the one construct that dropped its own state on the floor. +The fix (`checkFunction.switchStatement`) joins the ending state of every case that can fall off +its own end (skipping one that always `return`s or leaves the loop, the same distinction `if` +already draws) into the scope after the switch — nine lines, and the fixed case now proves +`Int[-512..1024]`, which matches what the interpreter really computes. + +### Fixed: a generated TypeScript `switch` fell through every case (Layer 3, TypeScript only) + +`seed 1 --count 10`, program `program::fuzz_2`; the interpreter and the TypeScript target disagreed +on every one of 3 cases in both idiom modes. `core/check.ts`'s `caseBody` correctly drops the +source's own case-closing `break` on the way into Core — documented in `docs/semantics.md` section +7, because the Core's `switch` never falls through and a target that printed the Core's cases as a +bare chain would read a kept `break` as leaving the enclosing loop instead. The TypeScript printer +(`targets/typescript/index.ts`) took that documented fact one step too literally: it printed each +case's body with nothing after it at all. JavaScript's `switch` *does* fall through without an +explicit `break`, so every case ran every case below it too — a single matching case pushed three +or four values into the accumulator instead of one. Go's `switch` does not fall through, Python is +compiled to an `if`/`elif` chain, and Rust's `match` arms don't fall through either, so this was +TypeScript-only. The fix appends an unconditional `break;` after every case and the `default` +(cheaper and just as sound as proving which bodies already exit on their own, since an unreachable +`break` after a `return` is harmless). `core`'s own source has no `switch` at all, so this had zero +effect on its committed output or conformance numbers — exactly the kind of gap only a generator +that writes `switch` reaches. + +### Fixed: an accumulator raised to a `const fold` was reassigned again later in the same function (Layers 1 and 3, every target) + +`seed 2 --count 50`, program `program::fuzz_20`; the TypeScript target threw +`TypeError: Assignment to constant variable.` at run time (a genuine Core-level bug, not a +TypeScript-only one — see below). Minimal form: + +```ts +export function f(p1: List, 3, 3>): Int { + let acc: Int = 10; + for (const e of p1) { + acc += e; + } + for (let i = 0; i < p1.length; i++) { + acc -= p1[i]; + } + return acc; +} +``` + +`optimize/optimize.ts`'s `raiseLoops` turns `let acc = init; for (x of xs) { acc = update; }` into +`const acc = seq.fold(xs, init, (acc, x) => update)` — sound on its own, and a nice simplification +every backend can choose to print as a native fold. It decides this by looking only at the +statement immediately after the loop; it never checked whether `acc` is assigned again *anywhere +later* in the function. Here it is: the second loop still reassigns `acc`. The pass raised the +first loop anyway, declared `acc` immutable, and left the second loop's plain `acc -= p1[i]` +untouched — a `const` two lines away from its own reassignment. This is a Core-level pass, so it +was not really "TypeScript's bug": Rust would have refused to compile it outright ("cannot assign +twice to immutable variable"), and Python and Go would likely have run it "successfully" while +silently keeping the wrong (unfolded-and-never-updated-again) value in the const case, which is +worse. The fix (`isReassignedLater`) walks the rest of the function body — including inside a +nested `if`, loop or `switch` — before raising, and skips the transform if `acc` is reassigned +anywhere in it. + +### Reported, not fixed: an unused `for…of` binding produces invalid Go + +`seed 2 --count 50`, program `program::fuzz_0` (one of several in that batch); Go's `go vet` refuses +the whole file with `declared and not used: e`. Minimal form: + +```ts +export function f(p1: List, 3, 3>): Int { + let acc: Int = 0; + for (const e of p1) { + acc += 1; + } + return acc; +} +``` + +The subset allows a `for…of` whose body never touches the bound element — nothing in +`docs/semantics.md` requires it, and a hand-written source could easily do the same thing on +purpose (counting elements, say). `targets/go/index.ts`'s `forEach` printer always names the +binding (`for _, e := range …`); Go requires every named local to be used, so an untouched one is a +compile error, and it takes down every function in the same generated file, not just the one that +has it. The natural fix is for the Go backend to print `_` when the loop body never references the +name. That check has to walk the *printed* Target AST rather than the Core's, because the printer +can still legitimately choose a lowering that references the loop variable in text the Target AST +does not model as a node (`adding-a-target.md`'s warning about `raw` emissions hiding what is +inside them from every later pass) — a correct fix has to either prove no such `raw` text exists in +the body or fall back to "assume used" whenever it might, and getting that exactly right was not a +small enough change to make confidently in the time this task had. Left as found, with this +reproduction and `node scripts/fuzz.ts full --seed 2 --count 1 --targets go` to replay it. + +### Not an engine bug: the generator's own nesting depth had an off-by-one + +Also found at scale, and worth recording because it looked exactly like a hang at first: two +loop-nesting parameters (`generate.ts`'s "one list per function may be 'wide' enough to force the +64-iteration widen/clamp path" and "loops nest up to two deep") were each safe in isolation but +compounded once they landed in the same function. A budget meant to allow at most two nested loops +let a third one through (the budget was spent only when *choosing* to nest, not when a loop was +actually created), so a generated program could range three loops deep, all three over the same +"wide", roughly-90-element list — about 90³ ≈ 730,000 executions of a body that pushes onto a list +with `push`'s copy-the-whole-list-per-call semantics (`interp.ts`), which is quadratic in the +list's own length. That combination made one generated program's *interpretation* — not its +compilation — take minutes, which is what made an early full run of `fuzz fast` look hung rather +than slow. Fixed in the generator (one extra `- 1`, `generate.ts`), not in the engine: nothing the +checker or the interpreter proved was wrong, the reference interpreter's `push` is simply not +written for the input sizes three-deep nesting of a "wide" list produces, and no hand-written +utility's loop nests three deep over a 90-element list today. Recorded here rather than silently +fixed because it shaped the generator's own depth and width defaults, which a reader tuning them +later should know about. diff --git a/engine/docs/intrinsics.md b/engine/docs/intrinsics.md new file mode 100644 index 000000000..68facf45b --- /dev/null +++ b/engine/docs/intrinsics.md @@ -0,0 +1,177 @@ +# Intrinsics + +Generated by `npm run docs`. One row per operation the Core can express; everything else is +source library code, which is the default answer (see the admission rule in +[semantics.md](semantics.md)). + +There are 104 intrinsics in 13 modules. + +## `clock` + +| operation | effects | comptime | meaning | +| --- | --- | --- | --- | +| `clock.durationMillis` | Pure | yes | The millisecond count of a duration. | +| `clock.elapsed` | Pure | yes | The duration between two instants, `to - from`, clamped at zero. | +| `clock.millis` | Pure | yes | A duration from a count of milliseconds. | +| `clock.now` | Clock | no | The current instant, in milliseconds since the Unix epoch. | +| `clock.sleep` | Clock | no | Suspends for a duration. Under the reference model this advances virtual time instantly. | + +## `core` + +| operation | effects | comptime | meaning | +| --- | --- | --- | --- | +| `core.eq` | Pure | yes | Structural equality. Both sides must have the same shape. | + +## `date` + +| operation | effects | comptime | meaning | +| --- | --- | --- | --- | +| `date.addDays` | Pure | yes | Shifts by a number of days, or `none` when the result leaves years 1 to 9999. | +| `date.clampEpochDays` | Pure | yes | A civil date from days since 1970-01-01, clamped into years 1 to 9999. Total, so it needs no Option. | +| `date.compare` | Pure | yes | Chronological comparison: -1, 0 or 1. | +| `date.day` | Pure | yes | The day component of a civil date. | +| `date.dayOfWeek` | Pure | yes | ISO day of week: Monday is 1 through Sunday is 7. | +| `date.diffDays` | Pure | yes | Exact day difference: `a - b`. | +| `date.fromEpochDays` | Pure | yes | A civil date from days since 1970-01-01, or `none` outside years 1 to 9999. | +| `date.fromYmd` | Pure | yes | A civil date, or `none` when the components do not name a real day. Never rolls over. | +| `date.isLeapYear` | Pure | yes | Whether a year has 366 days in the proleptic Gregorian calendar. | +| `date.month` | Pure | yes | The month component of a civil date. | +| `date.toEpochDays` | Pure | yes | Days since 1970-01-01. | +| `date.year` | Pure | yes | The year component of a civil date. | + +## `dec` + +| operation | effects | comptime | meaning | +| --- | --- | --- | --- | +| `dec.abs` | Pure | yes | Absolute value, keeping the scale. | +| `dec.add` | Pure | yes | Exact decimal add; both operands share a scale. | +| `dec.compare` | Pure | yes | Exact comparison: -1, 0 or 1. | +| `dec.divRound` | Pure | yes | Division to an explicit scale under an explicit rounding mode. | +| `dec.fromFloat` | Pure | yes | Rounds the exact binary64 value to a constant scale under an explicit rounding mode. | +| `dec.fromInt` | Pure | yes | An exact decimal from an integer, at a constant scale. | +| `dec.fromScaled` | Pure | yes | A decimal from its unscaled integer and a constant scale: fromScaled(1234, 2) is 12.34. | +| `dec.isNegative` | Pure | yes | Whether the value is strictly below zero. | +| `dec.mul` | Pure | yes | Exact decimal multiplication; the result's scale is the sum of the operands' scales. | +| `dec.rescale` | Pure | yes | Changes the scale under an explicit rounding mode. | +| `dec.sub` | Pure | yes | Exact decimal sub; both operands share a scale. | +| `dec.unscaled` | Pure | yes | The unscaled integer: dec.unscaled(12.34 at scale 2) is 1234. | + +## `float` + +| operation | effects | comptime | meaning | +| --- | --- | --- | --- | +| `float.add` | Pure | yes | IEEE-754 binary64 add, correctly rounded in every target. | +| `float.div` | Pure | yes | IEEE-754 binary64 div, correctly rounded in every target. | +| `float.fromInt` | Pure | yes | Exact conversion of an integer whose range fits binary64 without rounding. | +| `float.ge` | Pure | yes | Float comparison (ge). | +| `float.gt` | Pure | yes | Float comparison (gt). | +| `float.le` | Pure | yes | Float comparison (le). | +| `float.lt` | Pure | yes | Float comparison (lt). | +| `float.mul` | Pure | yes | IEEE-754 binary64 mul, correctly rounded in every target. | +| `float.neg` | Pure | yes | Float negation, which preserves the sign of zero. | +| `float.sub` | Pure | yes | IEEE-754 binary64 sub, correctly rounded in every target. | + +## `http` + +| operation | effects | comptime | meaning | +| --- | --- | --- | --- | +| `http.request` | Http | no | Performs one request. A transport error or a timeout answers `none`; a 4xx or 5xx status is a value, not a failure. | + +## `int` + +| operation | effects | comptime | meaning | +| --- | --- | --- | --- | +| `int.abs` | Pure | yes | Absolute value, whose range the checker proves from the operand's. | +| `int.add` | Pure | yes | Exact integer add. | +| `int.div` | Pure | yes | Truncated integer division. The divisor must be proven non-zero. | +| `int.ge` | Pure | yes | Integer comparison (ge). | +| `int.gt` | Pure | yes | Integer comparison (gt). | +| `int.le` | Pure | yes | Integer comparison (le). | +| `int.lt` | Pure | yes | Integer comparison (lt). | +| `int.max` | Pure | yes | The max of two integers. | +| `int.min` | Pure | yes | The min of two integers. | +| `int.mod` | Pure | yes | Remainder with the sign of the dividend. The divisor must be proven non-zero. | +| `int.mul` | Pure | yes | Exact integer mul. | +| `int.neg` | Pure | yes | Integer negation. Exact, like every integer operation: there is no wraparound. | +| `int.sub` | Pure | yes | Exact integer sub. | + +## `opt` + +| operation | effects | comptime | meaning | +| --- | --- | --- | --- | +| `opt.isNone` | Pure | yes | Whether the option is absent. This is what `x === undefined` means in the Core. | +| `opt.orElse` | Pure | yes | The value, or a default when absent. This is what `??` means in the Core. | +| `opt.some` | Pure | yes | Wraps a present value. | +| `opt.unwrap` | Pure | yes | The value inside an option the checker has proven present. | + +## `random` + +| operation | effects | comptime | meaning | +| --- | --- | --- | --- | +| `random.nextU32` | Random | no | A uniform 32-bit value. Everything derived from it (ranges, shuffles) is written in source, so the algorithm is identical in every target. | + +## `re` + +| operation | effects | comptime | meaning | +| --- | --- | --- | --- | +| `re.retain` | Pure | yes | Keeps only the scalars matching a comptime character class, dropping everything else. The class also refines the result. | +| `re.test` | Pure | yes | Whether the whole string matches the pattern. Always a full match; there are no partial matches in the subset. | + +## `seq` + +| operation | effects | comptime | meaning | +| --- | --- | --- | --- | +| `seq.all` | Pure | yes | True when every element satisfies the predicate. | +| `seq.any` | Pure | yes | True when at least one element satisfies the predicate. | +| `seq.at` | Pure | yes | The element at an index, or `none` when the index is outside the list. The checked form of seq.get. | +| `seq.concat` | Pure | yes | Concatenation of two lists. | +| `seq.contains` | Pure | yes | Whether a structurally equal element is present. | +| `seq.filter` | Pure | yes | Keeps the elements a pure predicate accepts. | +| `seq.find` | Pure | yes | The first element satisfying the predicate, or `none`. | +| `seq.fold` | Pure | yes | Left fold with an explicit initial value. | +| `seq.get` | Pure | yes | Element at an index proven to be in range. | +| `seq.indexOf` | Pure | yes | Index of the first structurally equal element, or -1. | +| `seq.len` | Pure | yes | Number of elements. | +| `seq.map` | Pure | yes | Applies a pure function to every element. | +| `seq.reverse` | Pure | yes | The list in reverse order. | +| `seq.slice` | Pure | yes | A slice, clamped to the list's length, from inclusive to exclusive. | +| `seq.sortStable` | Pure | yes | Stable sort by an explicit comparator returning a negative, zero or positive Int. | +| `seq.sortStableBy` | Pure | yes | Stable sort by a key: Int, Decimal, CivilDate or a string compared in scalar order. | +| `seq.sum` | Pure | yes | Sum of a list of integers or floats. | + +## `str` + +| operation | effects | comptime | meaning | +| --- | --- | --- | --- | +| `str.asAscii` | Pure | yes | Checked conversion: `some` when every scalar is below 0x80. | +| `str.asciiLower` | Pure | yes | ASCII-only lower casing: A-Z map to a-z, every other scalar is left alone. Defined on any string; a proven-ASCII argument unlocks the host's own case mapping. | +| `str.asciiUpper` | Pure | yes | ASCII-only upper casing: a-z map to A-Z, every other scalar is left alone. Defined on any string; a proven-ASCII argument unlocks the host's own case mapping. | +| `str.asDigits` | Pure | yes | Checked conversion: `some` when every scalar is an ASCII digit. | +| `str.charAt` | Pure | yes | The one-scalar string at an ASCII position. The index must be proven in range. | +| `str.charAtOpt` | Pure | yes | The one-scalar string at an ASCII position, or `none` when the index is outside. The checked form of str.charAt. | +| `str.codeAt` | Pure | yes | Code point at an ASCII position. The index must be proven in range. | +| `str.codeAtOpt` | Pure | yes | Code point at an ASCII position, or `none` when the index is outside. The checked form of str.codeAt. | +| `str.codePoints` | Pure | yes | The scalars of the string, as code points. | +| `str.compare` | Pure | yes | Scalar-order comparison: -1, 0 or 1. Never the host's collation. | +| `str.concat` | Pure | yes | Concatenation. Also the lowering of `+` on strings. | +| `str.contains` | Pure | yes | Whether `needle` occurs in the string. | +| `str.endsWith` | Pure | yes | Whether the string ends with `suffix`. | +| `str.fromCodePoints` | Pure | yes | Builds a string from code points. | +| `str.fromInt` | Pure | yes | Decimal representation of an integer, with a leading '-' when negative. | +| `str.indexOf` | Pure | yes | Scalar index of the first occurrence of `needle`, or -1. | +| `str.join` | Pure | yes | Joins a list of strings with a separator. | +| `str.len` | Pure | yes | Number of Unicode scalars in the string. | +| `str.padStart` | Pure | yes | Left pads with `pad` (one scalar) until the string has `length` scalars. | +| `str.parseInt` | Pure | yes | Parses an unsigned decimal integer; `none` when the string is empty or not all digits. | +| `str.repeat` | Pure | yes | The string repeated `count` times. | +| `str.slice` | Pure | yes | A slice of an ASCII string, clamped to its length, from inclusive to exclusive. | +| `str.split` | Pure | yes | Splits on a one-scalar ASCII separator. | +| `str.startsWith` | Pure | yes | Whether the string starts with `prefix`. | +| `str.trim` | Pure | yes | Removes leading and trailing whitespace, using the 25 code points JavaScript trims. | + +## `task` + +| operation | effects | comptime | meaning | +| --- | --- | --- | --- | +| `task.race` | Pure | yes | Runs idempotent tasks concurrently and takes the first one to answer `some`, or `none` when none does. | + diff --git a/engine/docs/progress.md b/engine/docs/progress.md new file mode 100644 index 000000000..18a160f06 --- /dev/null +++ b/engine/docs/progress.md @@ -0,0 +1,257 @@ +# Progress + +Status of the milestones and the measured metrics. Every number here is produced by +`node scripts/metrics.ts ../core`, `node ../engine/scripts/verify.ts .` or +`npm run conformance` / `conformance/bench.ts` in `core/`, and can be re-derived. + +## Milestones + +| Milestone | State | Evidence | +|---|---|---| +| M0 survey, contracts, specification | done | [`core/docs/survey.md`](../../core/docs/survey.md), [`core/docs/contracts.md`](../../core/docs/contracts.md), [semantics.md](semantics.md), [decisions/](decisions) | +| M1 frontend, HIR, semantic checker | done | `src/frontend`, `src/hir`, `tests/subset.spec.ts` (19 rejected constructs), `tests/boundary.spec.ts` | +| M2 Core IR, linking, interpreter, comptime | done | `src/core`, `src/link`, `src/interp`, `src/comptime`, `--dump core` | +| M3 analysis | done | ranges, refinements, effects and capability threading live in the checker ([ADR 0003](decisions/0003-checker-lowers.md)); `tests/refinements.spec.ts` | +| M4 conformance infrastructure | done | `src/conformance`, `core/conformance`, capability fakes driven by one fixture file | +| M5 backend framework and TypeScript | done | `src/backend`, `src/targets/typescript`, `LOWERING.md`, `API.json`, `SOURCEMAP.json` | +| M6 Python and Go | done | `src/targets/python`, `src/targets/go` | +| M7 minimal `CivilDate` and `Decimal` | done | `date.*` and `dec.*` intrinsics, `stdlib/date.ts`, the holiday and currency pilots | +| M8 falsification report | done | [targets/rust-sketch.md](targets/rust-sketch.md), [adding-a-target.md](adding-a-target.md), [adding-a-utility.md](adding-a-utility.md), this file | + +Not done, and deliberately: a tree-shaking bundle gate (needs a bundler in the toolchain), a baked +`Dataset` type, `Map`/`Set`, discriminated unions, and recursion. + +## Metrics + +### 1. Target conditionals outside the backends — **0** + +Enforced, not asserted: `tests/boundary.spec.ts` reads every source file under `frontend`, `hir`, +`core`, `link`, `comptime`, `analysis`, `optimize`, `interp` and `intrinsics` and fails on an +import of `targets/` or a comparison against a target name. The same test proves only +`src/frontend/lower.ts` imports the parser. + +### 2. Lowering mix + +| target | native | library | portable | +|---|---|---|---| +| TypeScript | 304 | 0 | 2 | +| Python | 292 | 12 | 2 | +| Go | 277 | 27 | 2 | +| Rust | 250 | 54 | 2 | + +The portable selections are the interesting ones: `str.compare` on a value that is not proven +ASCII, in **every** target, and the calendar conversions. The "library" column measures each +standard library rather than the engine: Rust needs twice as many as Go because `std` has no +regex, no left pad, no checked index and no stable sort that clones. + +### 3. Backend size + +| part | lines | +|---|---| +| frontend + HIR | 1166 | +| Core (checker, IR) | 2984 | +| analysis, link, comptime, optimize | 716 | +| intrinsics | 1827 | +| interpreter | 353 | +| backend framework | 1482 | +| target: TypeScript | 949 | +| target: Python | 1062 | +| target: Go | 1292 | +| target: Rust | 2384 | + +Each backend is smaller than frontend + Core + analysis (4866), which is the shape the +architecture predicts: the expensive part is meaning, not syntax. + +### 4. Generated expansion per utility + +Source lines against generated lines, per target: + +| utility | source | TypeScript | Python | Go | +|---|---|---|---|---| +| `is-valid-cpf` | 28 | 25 | 20 | 26 | +| `is-valid-cnpj` | 30 | 27 | 23 | 28 | +| `format-cnpj` | 28 | 24 | 20 | 30 | +| `generate-cpf` | 32 | 30 | 24 | 28 | +| `generate-cnpj` | 31 | 31 | 23 | 27 | +| `get-holidays` | 54 | 41 | 36 | 37 | +| `is-business-day` | 32 | 35 | 26 | 32 | +| `get-address-info-by-cep` | 104 | 88 | 62 | 76 | +| `format-currency` | 31 | 27 | 23 | 39 | + +Every utility is within 3× its source in every target; most are *smaller* than their source, since +the source carries the prose that says why. 902 lines of source (including the engine's standard +library) produce 895 + 676 + 1060 lines across the three targets. + +### 5. Big-integer representations — **0** + +No utility needs `bigint` or `math/big`: every proven range fits the platform-safe domain. +16 loops had their accumulator ranges widened rather than proven exactly, and 98 assignments were +clamped back into the platform domain under the bounded-step rule (`docs/semantics.md`, "Loops and +widening"). Both are reported rather than hidden, because a widened range is usually a hint that +the source could carry a tighter annotation. + +### 6. Conformance — **4256/4256 in every target, in both idiom modes** + +| comparison | result | +|---|---| +| reference interpreter vs the published npm package | 4247/4247 (every case that can be reproduced offline) | +| TypeScript, idiomatic and `--no-idioms` | 4256/4256 | +| Python, idiomatic and `--no-idioms` | 4256/4256 | +| Go, idiomatic and `--no-idioms` | 4256/4256 | + +The nine cases the npm comparison leaves out are `getAddressInfoByCep`, whose published +implementation performs real requests, and the two generators, which have no deterministic +reference to compare against; all nine are compared between the interpreter and the three targets, +on scripted responses and on the reference generator. What the generators draw is then fed back +through `isValidCpf` and `isValidCnpj` in the same run, so three targets agreeing bit for bit on +an invalid document would still fail. + +### 7. Idiomaticity + +`tsc --strict --noEmit`, `python3 -m compileall` and `go vet` are clean over the generated output, +and the output is formatted by Prettier, `ruff format` and `gofmt`. `verify` runs all of it. + +Read `out/python/get_holidays.py` and `out/go/get-holidays.go` side by side: the same utility is a +`sorted(key=…)` in one and a `slices.SortStableFunc` in the other, from one source. + +### 8. Performance + +**The bar is 1.0x, not 1.5x.** `core/bench/README.md`'s original budget was "within 1.5x of the +handwritten implementation"; every row below is now judged against **equal or faster**, and 1.5x +survives only as a line a row used to be allowed to cross. Numbers below are from +`node core/bench/run.mjs`, the cross-language harness (`core/bench/`), which asks the same +question of every language against the implementation that language's community actually ships — +`brazilian-utils/{python,go,rust}` for Python, Go and Rust; this package's own `src/` for +TypeScript. + +| language | utility | ratio before this pass | ratio now | moved by | +|---|---|---|---|---| +| typescript | `getHolidays` | 1.77x | **0.93x** | `date.fromYmd` native candidate, below | +| typescript | `isBusinessDay` | 1.26x | **0.66x** | inherits `getHolidays`' fix | +| typescript | `generateCpf` | 1.44x | **1.06x** | call-site inlining, budget 6 | +| typescript | `generateCnpj` | 1.24x | **1.10x** | call-site inlining, budget 6 | +| python | `formatCurrency` | 2.91x | **2.66x** | `trunc_mod`/`trunc_div` inlined; ASCII-byte `codePoints`/`fromCodePoints` | +| python | `generateCpf` | 1.86x | **1.65x** | call-site inlining, budget 12 | +| python | `generateCnpj` | 1.99x | **1.89x** | call-site inlining, budget 12 | +| go | `formatCurrency` | 1.08x | **0.93x** | ASCII-byte `re.retain`, `codePoints`/`fromCodePoints` | +| rust | `isValidCpf` | 2.73x | **2.57x** | ASCII-byte `re.retain` (`keep_digits`) | +| rust | `isValidCnpj` | 1.43x | **1.35x** | inherits the same `re.retain` fix | +| rust | `formatCurrency` | 1.81x | **1.74x** | one-buffer `str.concatAll`; ASCII-byte `re.retain` | +| rust | `generateCpf` | 2.37x | **1.68x** | one-buffer `str.concatAll` (nine-digit chain) | +| rust | `generateCnpj` | 1.89x | **1.22x** | one-buffer `str.concatAll` (twelve-digit chain) | + +Every other row (`isValidCpf`/`isValidCnpj`/`formatCnpj` in every language, Go's `generateCpf`/ +`generateCnpj`) was already at or under 1.0x and stayed there. `core/bench/README.md`'s +"What the numbers actually showed" and "Rust" sections carry the full account, row by row, +including the three shapes behind every fix (a missing native candidate, a per-target inlining +budget, one allocation instead of a chain) and what remains over 1.0x with the reason it cannot +come down further inside this pass: Python's `formatCurrency`/`generateCpf`/`generateCnpj` and +Rust's `isValidCpf`/`formatCurrency`/`generateCpf`. + +**The missing candidate.** `civilDate` (`core/out/typescript/lib/civil.ts`) computed a day forward +and then verified it by decomposing the result back through three more floor-division-heavy +Hinnant functions — a round trip that was 78% of `getHolidays`' call. The round trip only answers +one question, "is `day` within the month it names", which a days-in-month table (28-31, with +February's leap adjustment) answers directly; a native TypeScript candidate for `date.fromYmd` now +does the table check plus the single forward computation, with no `new Date` involved at all — a +`Date`-based candidate was tried first and measured *slower* than the portable round trip it was +meant to replace (Date construction and its getters cost more than four Hinnant functions), which +is exactly why every claim in this document is a measurement, not an assumption from the shape of +the problem. + +**Per-target inlining, measured per target.** `randomDigit → randomBelow → env.nextU32()` is a +three-layer call chain run 9-12 times per `generateCpf`/`generateCnpj` call, in every language that +has it. `engine/src/optimize/inline.ts` splices an eligible callee's body into its call site +(parameters bound once each, an early `return` turned into an `Option` assignment plus a `break` +where it is inside a loop), run once per target from `generate` with that target's own budget — +because whether this is free is a target property, not a program property: + +- **Python: aggressive (budget 12).** CPython pays a full stack frame per call with no JIT to elide + it; inlining the whole chain (`random_digit`, `random_below`, and incidentally `cpf_check_digit`/ + `is_repeated`, both under the same budget) took `generateCpf` from 1.86x to 1.65x and `generateCnpj` + from 1.99x to 1.89x. The trade is real: `generate_cpf.py`'s body is now one long flattened + function rather than a chain of four-line helpers, and Python has no bundle-size constraint to + weigh that against, so the budget stayed at 12. +- **TypeScript: conservative (budget 6), measured against a larger one.** V8 already inlines a + small monomorphic call once it is hot; a budget of 1 (inlining only `randomDigit`'s single-`return` + body, not `randomBelow`'s rejection-sampling loop) left `generateCpf`/`generateCnpj` at 1.14x-1.25x + — barely moved, confirming the JIT was not the bottleneck by itself. A budget of 6 (`randomBelow`'s + own size) reached 1.06x/1.10x. **Bundle-size effect, measured**: `generate-cpf.ts` grew from 45 to + 408 lines (1796 to 14695 bytes) and `generate-cnpj.ts` from 55 to 519 lines (1959 to 18305 bytes); + the whole `core/out/typescript` tree grew from 156322 to 187760 bytes (+20%), partly offset by + `lib/random.ts` disappearing entirely (everything that called it now has it spliced in) and by + `std/date.ts` shrinking from the `date.fromYmd` fix above. This is the trade the task asked to be + stated, not hidden: a real speed win, paid for in bytes, for a row that was already close to 1.0x + before the pass and stayed close after. +- **Rust: none, and the reason is itself a finding.** A first attempt at budget 6 measured *worse* + code, not better: this pass has no notion of a Rust borrow (`docs/decisions/0010-*.md`) — it binds + every inlined parameter as an owned local, so an inlined call to a function that ADR 0010 had + proven could borrow its argument instead printed a `.to_owned()` at every splice, one full string + clone per digit. That is exactly the allocation ADR 0010 exists to avoid, and by more than the + call overhead this pass would have removed, so `inlineBudget` is deliberately absent from + `RUST_BACKEND` (`engine/src/targets/rust/index.ts`'s own comment there has the same account). A + follow-up `#[inline]` attribute hint (a much smaller ask: let rustc's own inliner decide) was + tried and measured to change nothing (`isValidCpf`: 22-25ms with or without it, indistinguishable + from run-to-run noise) — rustc was already making the same decision either way — so that was + reverted too rather than kept as an unproven change. +- **Go: not attempted.** Every Go row was already at or under 1.0x before this pass except + `formatCurrency`, which the allocation fix below closed without touching call structure. + +**Allocation.** Two shapes, one in Rust, one in three languages at once: + +- `re.retain` (`keep_digits`, `keep_alphanumeric`) walked `.chars()`/`strings.Map`/ + `[ord(c) for c in whole]`-style, decoding the whole input as Unicode scalars before ever testing + one. Every class this project retains is ASCII (digits, upper and lower case letters), and an + ASCII byte needs no decoding to be range-tested — a multi-byte scalar's bytes are all ≥ 0x80, so + each one fails an ASCII range test on its own exactly as the decoded scalar would have, meaning a + byte-wise scan is not an approximation, it is the same predicate for less work. Rust + (`String::from_utf8(bytes().filter(...).collect())`) and Go (a `[]byte` scan) both got this + candidate, gated on every retained range being ≤ 127; a non-ASCII-range class (none exist in this + project today) still falls back to the scalar-wise pass. Python's and Go's `str.codePoints`/ + `str.fromCodePoints` (`group_thousands`' `out` list, proven `IntRange<0,127>` by its own source + type) got the same treatment for the same reason. +- Rust's string assembly built a `+` chain (`a + b + c`) as nested `str.concat` calls, each + allocating and copying everything to its left — `format_currency`'s `prefix`/`sign`/`body` + assembly and every `concat2` in `random_cpf_base`'s nine-digit chain were exactly this. + `backend/lower.ts`'s `operation` now flattens a chain of three or more pieces into their leaves + before lowering, and hands them to a new `str.concatAll` op when a target declares one (Rust + only, today; every other target falls through to the unchanged pairwise path). Rust's + `str.concatAll` sizes one buffer once, from every piece's own length, and pushes each piece into + it — with each length-costing piece bound to a local first, so a piece that is itself a call + (`group_thousands(keep_digits(&whole))`) runs once, not twice for sizing and pushing both; a + single-character literal piece prints `.push('x')`, not `.push_str("x")`, for + `clippy::single_char_add_str`. + +That is the architecture behaving as designed: performance was recovered by changing what the +compiler emits, not by rewriting a utility. Conformance stayed 4256/4256 in every target, in both +idiom modes, throughout every fix in this section; `node engine/scripts/fuzz.ts fast` and `full` +both stayed clean. + +### 9. Marginal cost + +| | first pilot (`isValidCpf`) | a late pilot (`formatCurrency`) | +|---|---|---| +| source lines | 28 (+ 39 in `lib/`) | 31 (+ 55 in `lib/`) | +| new intrinsics | 21 (`str.*`, `int.*`, `re.test`) | 0 | +| compiler changes | the frontend, the checker, the Core, the interpreter, three backends | none | +| conformance | the harness itself | 38 cases | + +The fourth and fifth pilots (`getHolidays`, `isBusinessDay`) needed the `date.*` intrinsics and +the standard library's calendar; the sixth (`getAddressInfoByCep`) needed the capability plumbing +and two ADRs; the seventh needed nothing. The eighth and ninth (`generateCpf`, `generateCnpj`) +needed one more intrinsic, `random.nextU32`, and everything derived from it — a uniform value +below a bound, by rejection sampling — is written in the subset, because a modulo bias would +otherwise have to match digit for digit across three standard libraries to stay invisible. The +cost is front-loaded exactly where the thesis says it should be. + +## What this does not yet prove + +- **One project, one domain.** The engine is domain-neutral by construction and + `examples/generic` keeps it honest, but only Brazilian Utils has been ported. +- **Nine utilities out of 138.** [`core/docs/survey.md`](../../core/docs/survey.md) says 120 of + them need only features that exist today; the remaining 18 need `Map`/`Set`, discriminated + unions or Unicode normalization. +- **Four targets, and the falsification held.** Rust was the one meant to break the design, and + it did not: `std` only, no crate, 4256/4256, and both frictions the sketch predicted turned out + to be about `std` rather than about the Core. What is still untried is a language whose strings + are UTF-16, which is where `str.compare`'s precondition gets its real test. diff --git a/engine/docs/semantics.md b/engine/docs/semantics.md new file mode 100644 index 000000000..8d06999a0 --- /dev/null +++ b/engine/docs/semantics.md @@ -0,0 +1,540 @@ +# Semantics + +This document is the specification the engine implements. The reference interpreter implements +it, every backend preserves it, and every disagreement between the two is a bug in one of them — +never a "platform difference". + +Where a rule exists because two languages disagree, the disagreement is named. That is the point +of writing the rule down. + +--- + +## 1. Shape of the system + +``` +authoring source (restricted TypeScript) + │ frontend (swappable) + ▼ +Semantic HIR ── the frontend contract + ▼ +Core IR ── what the program means + ▼ +target backends ── how each language represents that meaning + ▼ +native source per language +``` + +Three invariants: + +1. the **source** expresses the implementation; +2. the **Core IR** expresses what the implementation means, independent of any language, including + the authoring language; +3. a **backend** decides how that meaning is represented in its target. + +| Layer | Owns | +|---|---| +| Compiler (frontend to Core) | semantics, types, refinements, effects, proofs, conformance | +| Backend | representation, idioms, standard library selection, allocation, syntax, formatting | +| DX (handwritten, per language) | public API, input coercion, user-facing types, docs | + +The engine generates the **core**, never the public API. A DX may accept +`formatCnpj(12345678000199)`; the generated core takes +`formatCnpj(value: String, options: FormatCnpjOptions)`. Converting one into the other is the DX's +job, and that split is what keeps the core free of JavaScript-shaped coercion. + +--- + +## 2. Semantic types + +| Type | Meaning | +|---|---| +| `Bool` | boolean | +| `Int[lo..hi]` | mathematical integer with a finite, statically proven range | +| `Float` | IEEE-754 binary64 | +| `Decimal` | exact decimal with a compile-time scale | +| `String[min..max]` | immutable sequence of Unicode scalars; `length` counts scalars | +| `Ascii`, `Digits` | refinements of `String` | +| `List[min..max]` | immutable list with a length range | +| `Option` | explicit absence, written `T \| undefined` | +| `Record` | nominal, immutable, declared with `type X = { … }` | +| `Enum` | string-literal union, `"a" \| "b"` | +| `CivilDate` | proleptic Gregorian date, years 1 to 9999, no zone | +| `Instant`, `Duration` | integer milliseconds | + +Deliberately absent: `UInt` (an `Int` with `lo ≥ 0` already says it), a run-time `Regex` (patterns +are compile-time only), `Calendar` and holiday tables (source library code), `Timezone` and +`Period` (deferred until a utility needs them). + +Not yet admitted, and rejected with a diagnostic rather than silently mistranslated: discriminated +unions of records, `Map` and `Set`, and recursion. + +### 2.1 Integers + +Integers are mathematical: no wraparound anywhere, and every value carries a proven range. + +- Ranges are tracked with arbitrary-precision integers inside the compiler. +- Every collection length and every index derived from one lies in `[0, 2^31 - 1]` + (`MAX_COLLECTION_LENGTH`), which is the documented platform limit shared by the three targets. + That bound is what keeps every range finite without asking authors to annotate lengths. +- `Int` with no annotation means the platform-safe domain, `±(2^53 − 1)`: the integers every + target represents exactly with its default integer type. +- `IntRange` pins a tighter range, and the checker proves the value stays inside it. +- `number`, ordinary TypeScript's own spelling, is accepted and its range is *inferred* rather than + declared — see "Inferring a bare `number`" below. + +A backend picks a representation it can prove safe for the range: + +| Target | Representation | +|---|---| +| TypeScript | `number` within ±(2^53 − 1), `bigint` otherwise | +| Python | `int` (already arbitrary precision) | +| Go | `int`, assumed 64-bit (see [targets/go.md](targets/go.md)) | + +`/` and `%` are **truncated**: the quotient rounds toward zero and the remainder takes the sign of +the dividend, which is what JavaScript, Go, Java and C# do. Python's `//` and `%` are floored, so +the Python backend selects a native lowering only when both operands are proven non-negative and +a generated helper otherwise. Division and `%` require a divisor proven non-zero. + +`Math.min`, `Math.max` and `Math.abs` on an Int are `int.min`, `int.max` and `int.abs`; `Math.trunc` +and `Math.floor` on an Int are the identity, since an Int is already exact. There is no `float.*` +counterpart yet — admitting one needs a second caller, section 8's admission rule — so the same +calls on a `Float` are `E_MATH_FLOAT` rather than a made-up lowering. + +#### Inferring a bare `number` + +Inside a function, `number`'s range is already inferred the moment it is written: a local needs no +annotation at all (`let sum = 0`), and a parameter or a return type is just two more places the +same abstract interpretation runs. The only question is where the contract at a function's own +boundary comes from, and there are exactly two sound answers, one per kind of function +(`check.ts`, `checkFunction`): + +- **Library code** (anything not a root-level exported utility) has no published contract of its + own — it is already checked once per call site (section 3, specialization). A `number` parameter + there simply starts at the platform-safe default, the same starting point `Int` is, and the + caller's own proven type is substituted in exactly the way it already is for an explicit `Int`. + Nothing new has to happen for this case; it falls out of specialization for free. +- **A utility** (an exported function in a module at the source root) is the published API, so + there is no call site to take a range from — the range has to come from the utility's own body. + Every read of a `number` parameter outside the test of an `if` or a ternary (the guard's own + condition proves nothing about a use that has not happened yet; what its *result* narrows for the + rest of the function is what counts) is tracked, and once the body is fully checked, the union of + what was actually proven at each of those reads becomes the parameter's published type. A + parameter that is only ever read at its unconstrained default — no guard ever narrowed it before + a real use — or that is never read at all, cannot be given a published range without assuming or + silently degrading to a checked, portable lowering, both of which are unsound or throw away the + performance this project exists for; the checker refuses instead, with `E_BARE_NUMBER` naming the + parameter, saying that nothing narrows it, and naming the guard shape to add, in ordinary + TypeScript, never this engine's vocabulary. + +A return type written as a bare `number` is unconditionally inferred from what the body computes, +for a utility exactly as much as for library code: there is no published ceiling to stay under, so +there is nothing to prove ahead of time and nothing to refuse. + +`Int`, `IntRange`, `Float` and `Decimal` remain exactly what they were: an explicit, +deliberate statement of the contract, never requiring a guard, because writing one *is* the proof. + +### 2.2 Loops and widening + +Every loop has a proven trip count: counted `for` loops and `for…of` only, which is why `while` is +outside the subset. The checker re-checks a loop body until the types of its mutable locals stop +growing: + +- when the trip count is at most 64, the fixpoint is exact, and an accumulator ends with the range + it really has; +- above that, ranges widen to the platform-safe domain, and the loop is counted in the metrics. + +A widened counter may then be *clamped* back into that domain on assignment, but only while one +step is at most 2^22: since a loop runs at most 2^31 − 1 times, a bounded step cannot leave the +domain. A step the analysis cannot bound is an error, not a clamp. + +### 2.3 Strings + +A `String` is a sequence of Unicode scalars, and `length` counts scalars. + +- Generic strings support iteration by scalar (`str.codePoints`, or `[...s]`), concatenation, + comparison and the intrinsics. They never support positional indexing, because "position" means + a UTF-16 code unit in JavaScript, a byte in Go and a code point in Python. +- `str.codeAt`, `str.charAt` and `str.slice` — written `s.charCodeAt(i)`, `s.charAt(i)`/`s[i]` and + `s.slice(a, b)` — are admitted only on `Ascii` (and therefore on `Digits`), where the index means + the same thing in all three targets and is O(1) in each. On a string that is not proven ASCII + these are `E_UTF16_POSITION`, not a silent, target-dependent lowering: JavaScript's position is a + UTF-16 code unit, Python's is a code point and Go's is a byte, and the three disagree above + U+007F, so there is nothing to lower to. `s[i]` additionally picks between the unchecked + accessor and its checked, Option-returning form depending on whether the index is proven in + range — the same choice `xs[i]` makes below, and for the same reason: JavaScript answers + `undefined` past the end, which only the checked form can mean. +- `str.trim` removes exactly the 25 code points JavaScript's `String#trim` removes. Python's + `str.strip()` also removes U+001C to U+001F and U+0085 and does not remove U+FEFF, so the Python + backend passes the cut set explicitly; Go's `strings.Trim` takes the cut set as an argument + anyway. +- Case mapping is ASCII-only: `str.asciiUpper` and `str.asciiLower` map `a-z` and `A-Z` and leave + every other scalar alone. Full Unicode case mapping is out of scope because the targets disagree + ("ß" uppercases to "SS" in JavaScript, Python, Java and Rust, and stays "ß" in Go's + `strings.ToUpper` and C#'s `ToUpperInvariant`). A proven-ASCII argument unlocks the host's own + case mapping, which is equivalent there; anything else uses the portable implementation. +- `str.compare` is scalar order. JavaScript's `<` compares UTF-16 code units, so an astral scalar + would sort *below* U+E000 there and *above* it in Python and Go; the TypeScript backend + therefore selects the native comparison only for proven-ASCII strings and the portable one + otherwise. +- Unicode normalization is out of scope. It depends on the host's UCD version, which differs + between runtimes and between Python releases. + +### 2.4 Decimal + +`Decimal` is an exact decimal whose scale is a compile-time constant, represented in every +target as its unscaled integer. + +- `+`, `-`, `*` and comparison are exact. Multiplication adds the scales. +- Division, rescale and conversion from a float name their scale **and** their rounding mode at + the call site. There is no hidden global context: Python's default 28-digit context and Java's + throwing `BigDecimal.divide` are exactly what this rule avoids. +- `dec.fromFloat` rounds the *exact binary value* of the double, which is why `1.005` at scale 2 + under `half-up` is `1.00`: the double is really 1.00499999999999989…. A host that formats + `1.005` as `1,01` (as `Intl.NumberFormat` does) is rounding the shortest decimal representation + instead, which is a different operation and belongs to the DX. + +### 2.5 Time + +`CivilDate` is a day on the proleptic Gregorian calendar, years 1 to 9999 — the range Python's +`date` and C#'s `DateOnly` share — represented everywhere as days since 1970-01-01. + +- `date.fromYmd` answers an `Option`: there is no implicit rollover. JavaScript's `new Date(y, m, + d)` is `E_HOST_DATE`, not a lowering to it: `Date`'s months are zero-indexed where `fromYmd`'s + are 1-12, an out-of-range component silently rolls over into the next one instead of answering + `undefined`, and the value is bound to a timezone that a civil date never has. +- `dayOfWeek` is ISO: Monday is 1 through Sunday is 7. +- Month arithmetic is not admitted. When it is, it will have to name its overflow policy, because + JavaScript's `setMonth` rolls over (31 January + 1 month is 3 March) while Java's `plusMonths` + and C#'s `AddMonths` clamp. +- `Instant` is milliseconds since the Unix epoch and `Duration` is an exact count of + milliseconds. An `Instant` and a `CivilDate` are never interchangeable: converting between them + needs a zone, which is deferred, so a host date is the DX's problem. + +--- + +## 3. Refinements + +Refinements are what make an idiomatic native lowering safe. A lowering declares preconditions +over facts; without the fact, the portable implementation is selected. + +Facts: `Ascii`, `Digits`, `Matches`, integer ranges, string and list length ranges. + +They enter in four ways: + +1. **annotations**: `Digits`, `AsciiOf<14>`, `IntRange<0, 9>`, `List`; +2. **flow-sensitive analysis**: after `if (!re.test(PATTERN, value)) return …`, the rest of the + scope knows `value` matches the pattern, and therefore its class and its length range; a + `value.length !== 14` guard narrows the length; a comparison narrows an integer on both sides; + `x === undefined` narrows an `Option`; +3. **checked conversions**: `str.asAscii` and `str.asDigits` answer an `Option`; +4. **intrinsic guarantees**, declared by each intrinsic's signature. + +A fact crosses a function boundary by **specialization**: a library helper is checked once per +distinct call-site argument type, so `digitAt(cpf, index)` can serve an 11-digit CPF and a +14-digit CNPJ without either caller losing the proof it already had. Specializations that compile +to the same code are folded back together by the linker, which is sound precisely because the +types they differ in are proofs, not run-time structure. + +A utility — an exported function in a module at the source root — always keeps its declared +signature, because that signature is the published contract. + +--- + +## 4. Effects and capabilities + +Effects are `Pure`, `Fail`, `Http`, `Clock` and `Random`. The last three are capabilities. + +- An author calls `http.request(…)`, `clock.now()` or `random.nextU32()` and never mentions an + environment. The compiler infers effects over the call graph and threads a capability record + into exactly the functions that transitively need one (`analysis/capabilities.ts`). +- Threading is an internal concern of the generated code, not something a caller of a utility + ever sees. An exported utility keeps exactly the signature its source module declares; a + capability parameter never reaches that signature (see "The public entry point vs. the + capability-taking seam" below). Threading between a utility's own internal helpers is still + visible in generated source — an implementation detail worth reading, not one worth hiding. +- Each target that can build one generates its own default environment from its standard library — + `fetch`, `urllib.request`, the system clock, OS randomness — as generated code, not a runtime + package. Go and Rust cannot: see below. +- `Http` answers an `Option`: a transport error or a timeout is absence, and a 4xx or 5xx status + is an ordinary value. Retry and fallback are then written as ordinary control flow, which the + subset can express without `catch` (see [ADR 0006](decisions/0006-http-is-an-option.md)). Its + shape does not match JavaScript's `fetch`, whose Promise-of-`Response` splits the status and the + body across two separate awaits and fails by rejecting rather than by answering absent, so a bare + `fetch(...)` call keeps the generic host-global diagnostic (`call http.request`) instead of a + lowering. +- `Random` offers only `nextU32`. Everything derived from it — a range by rejection sampling, a + shuffle — is written in source, so the algorithm and its bias are identical everywhere. + `Math.random()`, a float in [0, 1), is `E_MATH_RANDOM` rather than a scaled `nextU32`: the scaling + itself would have to round identically in every target to stay unbiased, which is exactly the + kind of thing this rule keeps out of the core. +- **Async is computed, not written.** The TypeScript backend makes a function `async` exactly when + it reaches `Http`, and awaits its calls; Python and Go emit blocking code. + +### 4.1 The public entry point vs. the capability-taking seam + +A utility whose effects reach `Http`, `Clock` or `Random` is, at the Core level, a function that +takes a capability record. Its *published* signature is not: the source declares +`getAddressInfoByCep(cep: string)`, not `getAddressInfoByCep(cep: string, env: Capabilities)`, and +a drop-in replacement for the package that utility comes from has to keep it that way (see +[ADR 0011](decisions/0011-public-entry-points-vs-capabilities.md) for the fuller reasoning). + +Where a target can build a default environment from its own standard library (TypeScript, Python), +`generate()` (`backend/generate.ts`, `splitCapabilityEntryPoints`) splits such a utility in two: + +- a **public wrapper**, under the utility's own name, with the source's exact signature and no + capability parameter. It builds nothing itself — it calls the seam below with a module-level + default, built once at load time, never per call. +- an internal **seam**, named so it reads as one (`generateCpfWith` in TypeScript, + `generate_cpf_with` in Python), that still takes the capability record. It is not part of the + utility surface `API.json` lists as a normal function — it is marked there, under `seams`, + precisely so a reader does not mistake it for one — but it stays reachable (`export`/no leading + underscore) because two things need to reach it from outside its own module: the wrapper, and + the differential conformance driver, which calls it directly to inject a fixture-backed fake + instead of the real environment. This is why the seam, not the wrapper, is what a target's + driver dispatch table names. + +Where a target cannot build a default without either reaching past its standard library or +fabricating one (Go, Rust — neither ships an HTTP client, a CSPRNG or a default `Capabilities` +value anywhere in the generated *library*; the fake either driver builds lives in the driver +binary, never in `coreout`/`core`), no wrapper is generated. The capability-taking function stays +the only entry point, under its own original name, and is marked the same way in `API.json`'s +`seams` list (`hasWrapper: false`) so a DX author sees plainly that this one utility, unlike the +rest, needs an environment passed in by hand. This is a real, reported gap from drop-in parity, +not one papered over with a fake standard-library capability that would behave differently from +what a caller's own environment actually does. + +### 4.2 Concurrency + +The only primitive is `task.race(tasks)`, admitted because looking a CEP up in several services at +once needs it. + +- Every task answers an `Option`; the race answers the first task that answers `some`, or `none`. +- Under the reference model — virtual clock, scripted Http latencies — "first" is virtual + completion time, ties broken by task index, so a race is compared deterministically. +- Cancellation is best effort and semantically unobservable: a losing task may run to completion + and its answer is dropped. Only idempotent work belongs inside a race. +- Lowerings: `Promise.any` in TypeScript, a `ThreadPoolExecutor` in Python, goroutines and a + channel in Go. + +--- + +## 5. Errors + +```ts +export class InvalidCepError extends DomainError {} + +throw new InvalidCepError("CEP inválido"); +``` + +Error classes have empty bodies. A `throw` becomes the effect `Fail`, and each target +represents it natively: exception classes in TypeScript and Python, `(T, error)` with typed errors +and `errors.Is` compatibility in Go. + +Bugs are not domain errors. Overflow, invalid indexing, division by zero and non-exhaustive +matches are proven impossible at compile time; there is no `catch` to fall back on. + +--- + +## 6. Regex + +A regex is a compile-time value, normalized into explicit code point classes. + +Rejected: lookaround, backreferences, lazy quantifiers, `.`, inline flags, unanchored patterns and +the shorthand classes `\d`, `\w`, `\s` and `\b` — JavaScript's `\d` is ASCII-only while Python's +matches every Unicode digit, and `\s` differs again. Write the class out; the compiler prints it +in each dialect's own syntax (`\uXXXX` in JavaScript and Python, `\x{…}` in RE2). + +Anchoring is supplied by the target: `^…$` in JavaScript (which are string anchors without the `m` +flag), `re.fullmatch` in Python, `\A…\z` in Go. + +The accepted subset is linear-time on backtracking engines as well, because there is no +alternation of overlapping classes under an unbounded quantifier. + +--- + +## 7. Subset + +**Rejected**, each with a diagnostic code and, where possible, the construct to write instead: + +`any`, `unknown`; `==`; truthiness of anything but `Bool`; `null`; `?.` and `??` outside an +`Option`; `this`, prototypes, getters and setters, classes with bodies; dynamic property access; +objects used as maps; escaping mutable values; closures that capture mutable locals; effectful +lambdas inside combinators; generators, custom iterators, `for…in`; `while`; generic `try`/`catch`; +host globals (`Date`, `Intl`, `JSON`, `fetch`, timers, `console`, and every `Math` member except the +handful section 7.1 admits); regex constructs outside section 6; recursion. A bare `number` used as +a utility's parameter is accepted, but still refused with `E_BARE_NUMBER` when the body never +narrows it before using it — section 2.1, "Inferring a bare `number`". + +**Allowed**: `const` and `let` with local mutation; `if`/`else`; counted `for`; `for…of`; +`break` and `continue`; `return`; `throw` of a declared domain error; `switch` over an `Enum` with +exhaustiveness; template literals and the ternary operator; immutable records; enums; `Option`; +pure module-level functions, including pure lambdas passed to combinators; `push` on a local list +and assignment to a local list element; `dataset`-style constant tables. + +`break` and `continue` leave the innermost loop, and the state they carry is part of that loop's +fixpoint: what a local holds when a `continue` is taken reaches the next iteration, and what it +holds when a `break` is taken reaches the code after the loop, so a range proven for a loop-carried +value covers every path out of the body. A `switch` case ends with a `break` that closes the case +and nothing else; a `break` anywhere else inside a case is rejected with `E_SWITCH_BREAK`, because +a target whose `switch` does not swallow it would read it as a loop break instead. + +Only locals are mutable. A list may be built with `push` and is frozen when it escapes its +construction scope, so aliasing behaves identically across Go slices, Python lists, Rust ownership +and JavaScript arrays. + +### 7.1 Idiomatic spellings + +Everything above is described in the `str`/`seq`/`re`/`int`/`dec`/`date`/`random`/`task` namespace +vocabulary because that is what the frontend recognized first. An author writes ordinary +TypeScript instead, and never has to learn that vocabulary: the frontend (`frontend/lower.ts`, +`core/check.ts`) recognizes the idiomatic spelling on the left below and lowers it to exactly the +Core the namespace spelling on the right already produced. Both spellings stay accepted — the +namespace forms are the older spelling, not a deprecated one — and `tests/idioms.spec.ts` asserts +the equivalence directly, by checking that the two sides of every row compile to the identical +Core, rather than testing each spelling's *behavior* separately. + +| ordinary TypeScript | namespace form | +|---|---| +| `s.charCodeAt(i)` | `str.codeAt(s, i)` | +| `s.charAt(i)` | `str.charAt(s, i)` | +| `s[i]` | `str.charAt(s, i)`, or `str.charAtOpt(s, i)` when the index is not proven in range | +| `s[i] ?? fallback` | `str.charAtOpt(s, i) ?? fallback`, always — see "`??` forces the checked accessor" below | +| `s[i]?.charCodeAt(0)` | `str.codeAtOpt(s, i)` | +| `s.slice(a, b)` | `str.slice(s, a, b)` | +| `s.trim()` | `str.trim(s)` | +| `s.padStart(n, c)` | `str.padStart(s, n, c)` | +| `s.length` | `str.len(s)` | +| `s.toUpperCase()`, `s.toLowerCase()`, on a proven-ASCII string | `str.asciiUpper(s)`, `str.asciiLower(s)` | +| `[...s]` | `str.codePoints(s)` | +| `String(n)`, `n.toString()`, for `n: Int` | `str.fromInt(n)` | +| `s.replace(/[^…]/g, "")` | `re.retain` on the un-negated class | +| `PATTERN.test(s)`, for a regex literal or constant `PATTERN` | `re.test(PATTERN, s)` | +| `xs[i]` | `seq.get(xs, i)`, or `seq.at(xs, i)` when the index is not proven in range | +| `xs[i] ?? fallback` | `seq.at(xs, i) ?? fallback`, always — see "`??` forces the checked accessor" below | +| `xs.length` | `seq.len(xs)` | +| `xs.push(v)` | (already the only spelling) | +| `Math.min(a, b)`, `Math.max`, `Math.abs`, for `Int` operands | `int.min`, `int.max`, `int.abs` | +| `Math.trunc(n)`, `Math.floor(n)`, for `n: Int` | (the identity; an Int is already exact) | + +`s[i]` and `xs[i]` each pick between the unchecked accessor and its checked, Option-returning form +by the same rule: the unchecked one when the index is proven in range, matching JavaScript's own +guaranteed-present case exactly, and the checked one otherwise, because JavaScript answers +`undefined` past the end where Go and Rust panic — the same reason the Core keeps both forms of +each accessor in the first place (section 3, "checked conversions"). That rule decides the +accessor only where a *value* is wanted; see the next paragraph for `?? fallback`. + +**`??` forces the checked accessor, regardless of provability.** `xs[i] ?? fallback` and +`s[i] ?? fallback` always pick `seq.at`/`str.charAtOpt`, even where the index is proven in range +and a bare `xs[i]` would have picked the unchecked accessor. Under `noUncheckedIndexedAccess`, +real TypeScript already types a bracket index `T | undefined` no matter what the checker can +prove about the index (`tsc` has no access to that proof), so writing `?? fallback` is the +author's own statement, in the language's own terms, that they want the absent case handled — +not a claim about provability the checker would otherwise have to second-guess. Using the +provability rule here instead would mean the *same source text*, `xs[i] ?? fallback`, silently +lowers to a different Core depending on a fact about `xs` the author cannot see from the call +site, which is exactly the kind of surprise this frontier is built to avoid. + +**`s[i]?.charCodeAt(0)` is `str.codeAtOpt`, the checked *numeric* accessor.** `s.charCodeAt(i)` +alone has no `??` form: it answers `NaN` past the end, not `undefined`, so `s.charCodeAt(i) ?? +fallback` would compile under real `tsc` but never actually take the fallback branch — a +respelling that changes behavior, which this checker does not admit (compare the `.replace` +refusal below). But `s[i]` alone already answers `undefined` past the end, and chaining +`?.charCodeAt(0)` onto it reads the one scalar's code point only when it is present: the exact +case split `str.codeAtOpt` makes, spelled in ordinary TypeScript. Only this literal shape is +recognized — a plain (non-optional) bracket index and a literal `0` — since that is what keeps +the translation total; anything else is `E_OPTIONAL_CHAIN`, whose message for this one shape +names the accepted spelling directly. + +**`s.toUpperCase()`/`s.toLowerCase()` are `str.asciiUpper`/`str.asciiLower` once `s` is proven +ASCII**, the same gate `requireAsciiPositional` already applies to `charCodeAt`/`charAt`/`slice` +(section 2.3): JavaScript's case methods run Unicode's full case-folding table, which touches +scalars outside ASCII that `str.asciiUpper`/`asciiLower` leave alone, and folds those scalars +differently by target besides. Restricted to a proven-ASCII argument, the two case methods and +the two intrinsics are the identical function, so the ordinary spelling is sound there and only +there; an unproven string is `E_UNICODE_CASE`. + +Several idiomatic forms carry JavaScript-specific meaning that has no equivalent in Go, Python or +Rust, and the checker says so instead of lowering them regardless: + +- **`s.charCodeAt(i)`, `s.charAt(i)`, `s[i]` and `s.slice(a, b)` on a string not proven ASCII** — + `E_UTF16_POSITION`. JavaScript's "position" is a UTF-16 code unit, Python's is a code point and + Go's is a byte, and the three disagree on every scalar above U+007F; section 2.3 has the + detail. +- **`s.toUpperCase()`/`s.toLowerCase()` on a string not proven ASCII** — `E_UNICODE_CASE`. Unicode + default case folding touches scalars an ASCII-only table leaves alone, and differs again by + target outside U+007F, the same shape of problem as `E_UTF16_POSITION` above. +- **`Math.random()`** — `E_MATH_RANDOM`. It is a float in [0, 1), and the Random capability offers + only an unbiased 32-bit `nextU32`; section 4 has the detail. +- **`new Date(y, m, d)` and its friends** — `E_HOST_DATE`. Zero-indexed months, silent rollover and + a timezone binding, none of which `date.fromYmd` reproduces; section 2.5 has the detail. +- **`Math.min`/`max`/`abs`/`trunc`/`floor` on a `Float`** — `E_MATH_FLOAT`. There is no `float.*` + counterpart yet (section 2.1); a Float operand needs the comparison or the truncation written in + source instead of a made-up lowering. +- **`[...xs, ...ys]` and every array spread other than `[...s]` on a single string** — + `E_ARRAY_SPREAD`. Combining lists is `seq.concat`, a different operation with a different name, + not something JavaScript's spread syntax can stand in for. +- **`.replace` in every shape but a global, empty-replacement match of one negated class** — + `E_REPLACE_UNSUPPORTED`. `.replace` runs JavaScript's own replacement algorithm — capture group + substitution, a callback, only the first match without `/g/` — which nothing else has to + reproduce identically; the one shape that *is* target-independent, dropping every scalar outside + a class, is the one the checker accepts. + +--- + +## 8. Admission rule for intrinsics + +An operation becomes an intrinsic only if all three hold: + +1. it cannot be expressed efficiently and idiomatically as source library code; +2. it has a precise specification, a reference implementation and vectors; +3. at least two utilities need it, or it is a prerequisite of an admitted intrinsic. + +The default answer is: write it in source. Check digits, Easter, holidays, business days, pt-BR +currency formatting and the JSON field reader behind the CEP lookup are all library code. + +Every intrinsic that some target cannot lower natively with proven equivalence has a portable +implementation in the engine's own source-language standard library (`stdlib/`), compiled like any +other module. That is what makes a new target complete as soon as its core constructs lower: +native lowerings are optimizations, not requirements. + +--- + +## 9. Lowering selection + +Each target declares, per intrinsic, the candidates it has, what each requires, and what each +costs. Selection is deterministic and applied lexicographically: + +1. keep the candidates whose preconditions hold (facts, language baseline, allowed dependencies); + if none remain, it is a compile error; +2. prefer the lower declared cost class (allocations, then time complexity); +3. prefer `native`, then `library`, then `portable`; +4. break ties by declaration order. + +There is no measured tuning: costs are declared. Every non-trivial selection is written to +`out//LOWERING.md` with the rule that decided it, and that file is reviewed in pull +requests. + +--- + +## 10. Verification + +- **Layer 0 — random program generation.** A generator (`src/fuzz/`, `docs/fuzzing.md`) produces + well-typed programs in the subset above, biased toward loops with `break`/`continue`, nested + loops, a `switch` inside a loop, and indices derived from a loop counter — the shapes the two + soundness bugs found in this codebase both lived in. It runs the reference interpreter's actual + answers against the checker's own proven bounds (fast mode, wired into `verify`) and, on demand, + the interpreter against all four targets in both idiom modes (full mode), reusing the Layer 3 + machinery below. This is the layer that goes looking for a bug nobody wrote a test for yet; the + fixed 64 cases and two translation-validation programs are what confirms a known one stays fixed. +- **Layer 1 — reference semantics.** The interpreter is compared against the published package on + inputs inside the core's domain. Every divergence is classified as an interpreter bug, as + behavior owned by the DX, or as a documented semantic difference. +- **Layer 2 — intrinsic vectors.** Every intrinsic is checked against its own vectors, + independently of any utility. This is the layer that scales. +- **Layer 3 — differential per utility.** The interpreter and each generated target answer the + same cases through one JSON protocol, in both idiom modes. +- **Translation validation.** For every Core pass, the interpreter runs the Core before and after + the pass and the results must be identical. +- **Boundaries.** Tests read the sources to prove that nothing after the HIR imports the parser + and nothing before the backends imports or branches on a target. +- **Determinism.** Output is byte-identical across runs; `verify` regenerates and diffs. diff --git a/engine/docs/subset-gaps.md b/engine/docs/subset-gaps.md new file mode 100644 index 000000000..5a3da34b5 --- /dev/null +++ b/engine/docs/subset-gaps.md @@ -0,0 +1,243 @@ +# What ordinary TypeScript the subset still refuses + +The goal this file tracks: someone writing a utility should write TypeScript the way they normally +would, and never learn this engine's vocabulary. Where that is not true yet, it is written down +here with the diagnostic it produces, so the distance is measured rather than remembered. + +Each row was produced by compiling the smallest ordinary-TypeScript program that uses the feature, +one feature per project so that one rejection cannot hide another. The probes live in +`examples/features/` once they pass; until then they are in this file. + +## Re-running the twelve probes + +`number` is no longer refused outright, so every probe was re-compiled: one minimal program per row +below, plus three more of the original twelve that were never itemized here because they already +compiled cleanly (they cover constructs from "Accepted today"). One of those three happened to use +a `number` field as well and used to be blocked by `E_BARE_NUMBER` alongside everything else; it +now compiles cleanly on its own, which is the eighth of the "eight of the twelve" this file already +named. Of the nine rows below, seven (every one of them except "default and optional parameters" +and "method call syntax", neither of which ever used a `number`) used to report `E_BARE_NUMBER` +alongside their row's own diagnostic — sometimes the only diagnostic that made it out at all, when +`E_BARE_NUMBER` came from the frontend and cut the compile short before the real gap was even +reached. All seven now report only their row's own diagnostic — each row below says so. None of +them compiles cleanly (inference removes the mask, not the underlying gap), so none moves to +`examples/features/` yet. + +One thing the re-run surfaced that is worth recording even though it has nothing to do with +`number`: "Method call syntax" below was stale. It no longer needs a recognizer — that landed in +an earlier change to this same subset — so the row has been rewritten to describe what it actually +hits now. + +## Refused today + +### `number` as a parameter or return type — accepted, with `E_BARE_NUMBER` refusing one case + +```ts +export function sumPoint(x: number, y: number): number { + return x + y; +} +``` + +This was the single biggest barrier: it was hit by eight of the twelve probes, masking whatever +else each one exercised. It no longer is — `number` is accepted exactly as an ordinary TypeScript +author writes it, and its range is inferred rather than demanded (`check.ts`, "Inferring a bare +`number`"; `docs/semantics.md` §2.1). `Int` and `IntRange` are still there for an author who +wants to state the contract explicitly, and neither ever needed a guard, because writing one *is* +the proof. + +Inside a function the range was already inferred, by the same abstract interpretation the loop +fixpoint runs — a local needs no annotation either (`let sum = 0` always worked). What was missing +was a contract at the boundary, and there are two places to find one without an annotation: + +- **The call sites** of a helper — anything not an exported function of a module at the source + root. The checker already specializes one of those per call site (ADR 0004), the same way + `isRepeatedRun` is specialized today; a `number` parameter there simply starts at the + platform-safe default `Int` already does, and the caller's own proven type is substituted in. + Nothing new was needed for this case. +- **A guard the author already writes**, for an exported utility, which has no call site of its own + to take a range from. `if (n < 0 || n > 99) return 0;` is a proof: every read of the parameter + outside a guard's own condition — the condition proves nothing about a use that has not happened + yet, only its *result* narrows anything — is tracked, and the union of what was actually proven + at each of those reads becomes the parameter's published type. + +What remains is exactly `sumPoint` above: an exported function whose body guards nothing before +using `x` and `y`. That is refused, with two diagnostics, one per parameter: + +``` +E_BARE_NUMBER: `x` is a bare `number`, and `sumPoint` never narrows it before using it + help: add a guard before x is used, for example `if (x < 0 || x > 99) return …;` — the range a + guard like that proves becomes x's published contract; write `x: Int` instead if the full range + really is what is meant +``` + +A parameter that is read nowhere at all is refused too, with its own message (`` `n` is never used, +so `f` proves nothing about its range ``), so an unused bare `number` is never silently accepted +either. Tests: `engine/tests/number-inference.spec.ts`. + +### `while` — `E_WHILE` + +```ts +while (index < n) { + total += index; + index++; +} +``` + +Refused because nothing bounds it, and every range the checker proves depends on a loop running a +knowable number of times. A counted `for` carries its own bound. To accept `while`, the condition +has to yield one — a variable that provably moves toward the bound by at least a fixed step — or +the loop needs a declared ceiling. Reachable, and the most valuable single item after `number`. + +Re-run: a full probe (`n: number`, `while (index < n) { … }`) used to report `E_BARE_NUMBER` twice +alongside `E_WHILE`, all three from the frontend in one pass. It now reports `E_WHILE` alone. + +### Destructuring — `E_DESTRUCTURING` + +```ts +const { x, y } = point; +const [first, second] = pair; +``` + +Pure syntax. It desugars to field reads the subset already has, and nothing about it is hard; it +is refused only because the frontend never learned it. + +Re-run: both the object and the array form (`{ x, y }` off a record with `number` fields; `[first, +second]` off a `number[]`) used to report two or three `E_BARE_NUMBER`s alongside `E_DESTRUCTURING`. +Both now report `E_DESTRUCTURING` alone. + +### Default and optional parameters — `E_PARAM_PATTERN`, `E_SIGNATURE` + +```ts +export function greet(name: string, upper: boolean = false): string; +export function maybeSuffix(value: string, suffix?: string): string; +``` + +A default desugars to a conditional at entry. An optional parameter is `T | undefined`, which the +Core already models as `Option`. Both are close to free and both are what a normal signature looks +like. + +Re-run: neither probe used `number`, so neither was affected by the inference work, and `greet` +still hits `E_PARAM_PATTERN` exactly as before. `maybeSuffix` turned out to need its own probe +correction: the frontend's `params()` never reads a parameter's `?`, so `suffix?: string` is not +rejected at all today — it is silently read as a required `string`, and the body's own +`suffix === undefined` then fails with `E_UNDEFINED_COMPARE` (a `String` is never `undefined`) +rather than with the diagnostic this row names. That mistranslation is a real, separate gap this +row should track going forward, and is unrelated to `number`, so it is left unfixed here. + +Confirmed directly against the parser rather than inferred from the diagnostic: `b?: string` +parses as an `Identifier` carrying `optional: true`, and `params()` reads `name` and +`typeAnnotation` and nothing else, so the flag is dropped on the floor. `c: boolean = false` +parses as an `AssignmentPattern`, which `params()` does reject (`E_PARAM_PATTERN`). So of the two +halves of this row, one is a loud refusal and the other is a **silent miscompile** — the worst +shape a gap can have, and the reason this row should be closed before the rest of the list. + +What both should become, when it is closed: an optional parameter is `Option`, which the Core +already models end to end, so the only new work is in the frontend. A default parameter keeps a +required parameter in the Core and substitutes the default expression at every internal call site +that omits the argument — the classic desugaring, which needs no `Option`, no renaming and no +change to the body, and which leaves the published signature saying what the author wrote. The +engine generates the core and never the public DX API (ADR 0011), so an exported utility's default +belongs to the wrapper, not to the generated signature. + +### Discriminated unions — `E_UNION`, `E_SWITCH_SUBJECT`, `E_MEMBER` + +```ts +type Shape = { kind: "circle"; radius: number } | { kind: "square"; side: number }; +``` + +The one genuine type-system feature missing. `core/docs/survey.md` counts this among what the +remaining 18 of 138 utilities need. It requires a sum type in the Core and a representation in each +target — a tagged struct in Go, an `enum` in Rust, a tagged dataclass in Python, a tagged object in +TypeScript — plus exhaustiveness on the tag, which the `switch` checker mostly has already. + +Re-run: `radius`/`side` are `number` fields, and used to hit `E_BARE_NUMBER` before the union itself +was ever considered. It now reaches `E_UNION` directly. + +### `Map` and `Set` — `E_NEW`, `E_METHOD`, `E_MEMBER` + +```ts +const seen = new Set(); +``` + +Also counted in the survey's remaining 18. The obstacle is not the data structure, it is that +**iteration order has to mean the same thing everywhere**: JavaScript's `Map` iterates in insertion +order, Go's map iteration is deliberately randomized, Python's `dict` is insertion-ordered, and +Rust's `HashMap` is unordered. Supporting it means picking insertion order and generating it, not +mapping onto whatever each language calls a map. + +Re-run: a probe returning the `Set`'s size as `number` used to hit `E_BARE_NUMBER` on that return +type before the checker ever reached the `new Set()` call. It now reaches `E_NEW` directly. + +### Recursion — `E_RECURSION` + +```ts +export function factorial(n: number): number { + return n <= 1 ? 1 : n * factorial(n - 1); +} +``` + +Refused because nothing bounds the depth, and a generated Rust or Go program has a real stack. +Two ways out, both partial: self-tail-recursion can become a loop mechanically, and a depth the +checker can bound makes the rest safe. General recursion with an unbounded depth stays out. + +Re-run: `n` used to hit `E_BARE_NUMBER` before the call graph was even walked. It now reaches +`E_RECURSION` directly — the ternary's own `n <= 1` test narrows `n` to `Int[2..9007199254740991]` +for the recursive branch, which is enough of a guard on its own, so nothing about `number` stands +in the way here at all. + +### `try`/`catch` — `E_TRY` + +```ts +try { + return Number.parseInt(value, 10); +} catch { + return 0; +} +``` + +The one where the languages genuinely disagree. Rust has no exceptions, Go has no `try`, and the +Core already expresses failure as an effect (`Fail`) that becomes `Result` in Rust, a second +return value in Go, and a raise in Python and TypeScript. A narrow `try`/`catch` around a call that +fails could map onto that. Arbitrary `try` around arbitrary code cannot, and should keep being +refused with a diagnostic that names the shape that works. + +Re-run: the declared return type is `number`, and used to report `E_BARE_NUMBER` alongside `E_TRY` +(the return type is processed before the body). It now reports `E_TRY` alone. + +### Method call syntax — recognized; the remaining refusals are the ordinary ASCII proof rule + +```ts +value.split("-").join(" ").toUpperCase().trim() +``` + +Not a missing capability, and no longer a missing recognizer either: `check.ts`'s `methodCall` +reads `.split`, `.join`, `.toUpperCase`, `.trim` and the rest of the ordinary method spellings +directly, each onto the intrinsic it already names. + +Re-run: unaffected by this work (no `number` anywhere in the probe), but the row itself was stale — +this probe no longer hits `E_METHOD` for lacking a recognizer. It hits `E_UNICODE_CASE` on +`.toUpperCase()`, because `value` is a plain `string`, not proven ASCII, and the two case-mapping +tables genuinely disagree above U+007F (§2.3), the same rule a direct `str.asciiUpper` call would +also need satisfied. A version of the probe starting from an already-ASCII value compiles cleanly. + +## Accepted today + +`??` on an `Option`, counted `for`, `for…of`, `break`, `continue`, early `return`, `switch` over an +enum with exhaustiveness, template literals, the ternary operator, immutable records, `push` on a +local list, pure lambdas passed to combinators, `throw` of a declared domain error. + +`.filter()`, `.map()` and `.reduce()` are accepted as shapes. `.filter()` over a `number[]` now +compiles cleanly end to end: the element type is `Int[default]`, the same unconstrained range `Int` +already had, since a list element is not a function boundary and has no guard or call site of its +own to narrow it from. `.map()` and `.reduce()` still fail on a `number[]` with unconstrained +elements, but not on the combinator or on `number` — doubling or summing an unconstrained element +can genuinely overflow the platform-safe domain, the same arithmetic-safety diagnostic `sumPoint` +above hits, so both still need either a guard on the elements or an explicit `IntRange` to bound +them, exactly as they would with an explicit `Int`. + +## How this file is meant to end + +Every row moves to "accepted", or stays with a diagnostic that tells an author what to write +instead — in ordinary code, naming the construct and why it has no equivalent, never in this +engine's vocabulary. A row that stays is a fact about the four languages, not a gap in the engine, +and it has to read that way to the person who hits it. diff --git a/engine/docs/targets/go.md b/engine/docs/targets/go.md new file mode 100644 index 000000000..4c911c11b --- /dev/null +++ b/engine/docs/targets/go.md @@ -0,0 +1,60 @@ +# Target: Go + +**Baseline** Go 1.21. **Dependencies** standard library only. **Formatter** `gofmt`. +**Linters** `go vet`, `staticcheck`. + +## Representation + +| Semantic type | Go | +|---|---| +| `Int[lo..hi]` | `int`, **assumed 64-bit** | +| `Float` | `float64` | +| `Decimal` | `int` holding the unscaled integer | +| `String`, `Ascii`, `Digits` | `string` | +| `List` | `[]T` | +| `Option` | `*T` | +| `Record` | a struct with exported fields | +| `Enum` | `string` | +| `CivilDate`, `Instant`, `Duration` | `int` | + +`int` is 64 bits on every platform Go supports except the 32-bit ones. A range that leaves +±(2^53 − 1) is reported as a metric; a range that would leave 64 bits is a compile error, since no +`math/big` lowering is admitted yet. + +## Shape of the output + +- One package, `core`, with one file per source module and a `support.go` holding the generic + helpers the capability tables name (`ptr`, `at`, `orElse`, the sequence helpers, `raceFirstSome`). +- A utility is exported (`IsValidCpf`); library helpers stay unexported (`digitAt`). +- A `Fail` effect becomes `(T, error)`. A fallible call is hoisted into + `value, err := f(); if err != nil { return zero, err }`, which is what a Go author writes by + hand. +- Go has no conditional expression, so a `cond` is hoisted into a temporary and an `if`/`else`. +- Combinators become loops. +- Domain errors are structs in `errors.go` with `Error()` and `Unwrap()` to a package-level + `ErrDomain`, so `errors.Is` recognizes the family and `errors.As` recognizes the member. +- `race` is one goroutine per task and a buffered channel. +- **No default `Capabilities`.** `support.go` declares the interface only — no HTTP client, no + clock, no RNG anywhere in `core`, on purpose: the package imports nothing beyond what the + utilities' own logic needs. The differential driver's `cmd/driver/main.go` builds its own + fixture-backed fake, but that fake lives in the driver binary, never in the library. This means + a utility whose effects reach `Http`, `Clock` or `Random` (`GetAddressInfoByCep`, `GenerateCpf`, + `GenerateCnpj`) cannot get the public-wrapper treatment TypeScript and Python give the same + utilities (`docs/semantics.md` §4.1) — there is no default to hand a wrapper, and one is not + fabricated to manufacture the appearance of parity. The capability-taking form stays the only + entry point, under its original name, taking `Capabilities` as a normal parameter a caller + supplies. This is a genuine, reported gap from drop-in replacement, not a defect in this + backend's generation — see [ADR 0011](../decisions/0011-public-entry-points-vs-capabilities.md). + +## Notable lowerings + +`len(s)` counts bytes, so `str.len` is native only on proven-ASCII values; otherwise it is +`len([]rune(s))`, which costs an allocation and is declared as such in the cost table. + +`strings.Compare` compares UTF-8 bytes, which is code point order, so scalar-order comparison is +native here with no precondition — unlike TypeScript. + +RE2 has no backtracking and no `\uXXXX` escape: patterns are printed with `\x{…}` and anchored +with `\A…\z`. + +`slices.SortStableFunc` is stable; `sort.Slice` is not, and is never selected. diff --git a/engine/docs/targets/python.md b/engine/docs/targets/python.md new file mode 100644 index 000000000..af9eba034 --- /dev/null +++ b/engine/docs/targets/python.md @@ -0,0 +1,60 @@ +# Target: Python + +**Baseline** Python 3.9. **Dependencies** none. **Formatter** `ruff format`. +**Linters** `ruff check`, `pyright`. + +## Representation + +| Semantic type | Python | +|---|---| +| `Int[lo..hi]` | `int` (already arbitrary precision, so a range never forces a change) | +| `Float` | `float` | +| `Decimal` | `int` holding the unscaled integer | +| `String`, `Ascii`, `Digits` | `str` | +| `List` | `List[T]` | +| `Option` | `Optional[T]` | +| `Record` | a frozen `@dataclass` | +| `Enum` | `Literal["a", "b"]` | +| `CivilDate`, `Instant`, `Duration` | `int` | + +## Shape of the output + +- One module per source module, with relative imports; `_support.py` holds the generated + capability defaults, the engine's record types and the two division helpers. +- A module exports exactly what its source module exports, no more: a function the source never + wrote `export` on (`is_ok` alongside `get_address_info_by_cep`, say) is renamed with a leading + underscore, on its declaration and on every call site that names it — Python's own convention + for "not part of this module's surface" — and left out of the module's `__all__`, the second + half of that same convention. Every public function is still listed in `__all__` explicitly. +- Comprehensions for `map` and `filter`, `any`/`all` over a generator, `next(…, None)` for `find`, + `sum` for a sum, `sorted(key=…)` for a keyed sort — the idiomatic form in each case, with the + lambda inlined into the comprehension when it is a single expression. +- A fold becomes a loop: `functools.reduce` is not idiomatic Python. +- Domain errors are classes in `errors.py` extending a generated `DomainError(Exception)`. +- `race` is a `ThreadPoolExecutor` with `as_completed`. +- `_support.py`'s `Capabilities` class *is* the platform default (`urllib.request`, `time`, + `random`), not an interface with a separate implementation; `DEFAULT_CAPABILITIES` is one + instance of it, built once at import time. **A utility that reaches `Http`, `Clock` or `Random` + is a public wrapper over an internal seam** (`docs/semantics.md` §4.1, + [ADR 0011](../decisions/0011-public-entry-points-vs-capabilities.md)): `get_address_info_by_cep` + keeps the source's exact signature and calls `get_address_info_by_cep_with(cep, + DEFAULT_CAPABILITIES)`, a sibling in the same module the differential driver calls directly to + inject a fixture-backed fake. Neither gets the underscore treatment above — both stay in + `__all__` — but only the wrapper is listed as a utility in `API.json`; the seam is marked there + under `seams` instead. + +## Notable lowerings + +`len` counts code points, which *is* the Core's definition of length, so `str.len` is native with +no precondition — the opposite of TypeScript. Positional access still needs the ASCII proof, +because an index into a `str` is a code point index while the Core's is a scalar index into an +ASCII string. + +`//` and `%` are floored. The Core truncates, so the native operators are selected only when both +operands are proven non-negative; otherwise the generated `trunc_div` and `trunc_mod` are used. + +`str.strip()` uses Python's own whitespace set, which includes U+001C to U+001F and U+0085 and +excludes U+FEFF. `str.trim` therefore passes the 25 code points explicitly. + +`str.upper()` is ASCII-equivalent only on ASCII input, so it carries the same precondition as +TypeScript's `toUpperCase`. diff --git a/engine/docs/targets/rust-sketch.md b/engine/docs/targets/rust-sketch.md new file mode 100644 index 000000000..eba735032 --- /dev/null +++ b/engine/docs/targets/rust-sketch.md @@ -0,0 +1,110 @@ +# Sketch: what a Rust backend would look like + +**Carried out.** `engine/src/targets/rust/index.ts` exists, is registered as a fourth target, and +passes the same bar the other three do: `cargo build --offline --release`, `cargo clippy --offline +-- -D warnings` and `rustfmt --check` are clean, in both idiom modes, and the differential +conformance runner matches 4256/4256 cases against the reference interpreter — the same count the +other three targets report. See `docs/targets/rust.md` for the finished representation table and +notable lowerings, and [ADR 0009](../decisions/0009-rust-values-are-owned.md) for the one decision +below that turned out not to be obvious once real code was going through it. + +**Where reality differed from this sketch**, the most interesting result of carrying it out: + +1. **The two predicted frictions were exactly right.** No HTTP client in `std` → a generated + `Capabilities` trait, same shape as every other target's capability record. No regex in `std` → + a hand-written matcher, but not the "generated scanner" this sketch guessed at: a small general + NFA-style interpreter (`support.rs`'s `re_ends`) over a `static ReNode` tree built once per + pattern at compile time by rustc — closer to "one JSON codec, one regex codec" than to "one + generated function per pattern." Concretely faster than Go's current `re.test`, which recompiles + its pattern from source text on every call (measured during this task, on a benchmark run in + parallel with it): the Rust matcher never compiles anything at call time, because rustc already + placed the compiled `ReNode` data in the binary. +2. **`&str` parameters do not survive contact with the shared pipeline.** This sketch's ownership + plan ("parameters borrow, results own") assumed borrow decisions could be made locally, the way + a hand-written Rust function's author would make them. They cannot: a capability table's `emit` + runs at lowering time, before any function's scope exists to consult, and `printModule` sees one + module at a time with no access to another module's function signatures — neither of which this + sketch had reason to consider, because it was reasoning about one function, not about the + pipeline that emits the whole program before any one function's text is final. Every value is + owned instead ([ADR 0009](../decisions/0009-rust-values-are-owned.md)), at the cost of a few + more `.to_owned()` calls than a hand-tuned port would write and one Rust-specific rewrite + (`opt.unwrap` sometimes needs `.as_ref()` first) this sketch did not anticipate at all, because + the sketch's borrowed design would not have needed it — Go's own pointer-dereference version of + the same narrowed read has no such problem, since a Go pointer dereference is free to repeat. +3. **One flat `CoreError` enum, not one type per utility.** Simpler than sketched, and what makes + the shared Go-shaped fallible-call hoist collapse into a plain `f(...)?` at print time: there is + only ever one error type in the program, so `?` never has to bridge a mismatch. +4. **`Enum` is `String`, matching Go, not the sketch's `#[derive(Clone, Copy, PartialEq)]` + fieldless enum.** The project has two enum-shaped types (`CnpjVersion`, `HolidayType`) and + neither is ever matched with a `switch`, only compared with `===`, so there is no exhaustiveness + the richer representation would buy back today. `switch` over an `Enum` is still handled (a + `match`, with a defensive `_ => unreachable!()` arm, since matching on `String` cannot be + statically exhaustive the way matching on a real `enum` could be) — this is a case where a + richer, sketch-shaped representation remains the right answer if a project ever needs it, just + not yet demonstrated by one that does. +5. **`i64` throughout, exactly as sketched** — every proven range in this project fits the + platform-safe domain, so the sketch's `i128`/never-`num-bigint` ceiling was never exercised. If a + future range needs it, that is still the finding the sketch named. + +The rest of the sketch below is the original falsification exercise, kept as the record of what was +reasoned out *before* any Rust code existed to correct it. + +--- + +No Rust is generated yet. This is the falsification exercise the architecture asks for: walk every +Core construct and every intrinsic, and say how Rust would represent it — and where it would not +fit. + +## Representation + +| Core | Rust | +|---|---| +| `Bool` | `bool` | +| `Int[lo..hi]` | the narrowest of `i8`…`i64`/`u8`…`u64` that contains the range, `i128` beyond, `num-bigint` never (it is a dependency) | +| `Float` | `f64` | +| `Decimal` | `i64` or `i128` unscaled, chosen from the proven range | +| `String` | `&str` for parameters, `String` for results | +| `Ascii`, `Digits` | `&[u8]` behind a newtype, so indexing is a byte index and O(1) | +| `List` | `&[T]` for parameters, `Vec` for results | +| `Option` | `Option` | +| `Record` | a `#[derive(Clone, PartialEq)]` struct with public fields | +| `Enum` | a fieldless `enum` with `#[derive(Clone, Copy, PartialEq)]` | +| `CivilDate`, `Instant`, `Duration` | `i32`, `i64`, `i64` newtypes | + +## Constructs + +- **Ownership.** Every value in the Core is immutable once it escapes, and a mutable local never + escapes, so parameters borrow (`&str`, `&[T]`) and results own (`String`, `Vec`). The + frozen-on-escape rule is exactly what makes this mechanical; without it, borrow inference would + be the hard part of the backend. +- **`Fail`.** `Result` with a generated error enum per utility, `?` at the call sites the + lowerer already hoists for Go. The Go hoisting pass is reusable as is. +- **Combinators.** Iterator chains: `map`, `filter`, `fold`, `any`, `all`, `find`, and + `sort_by_key` for a stable sort (`sort_by_key` is stable; `sort_unstable_by_key` is not and is + never selected). +- **Loops.** `for i in a..b` and `for item in slice`, which is what the Core's two loop forms are. +- **Capabilities.** A trait parameter: `fn get_address(cep: &str, env: &E)`. The + default implementation would need a dependency for HTTP (`ureq`, `reqwest`), which the "no + external dependencies" rule forbids — so the default environment would have to be behind a + feature flag, or omitted with the DX supplying one. **This is the first real friction.** +- **Async.** Rust has no runtime in `std`. The async colouring the TypeScript backend applies has + no equivalent, so `race` would be threads and a channel (`std::sync::mpsc`), like Go without + goroutines being cheap. Acceptable for two concurrent GETs, wrong for hundreds. +- **Regex.** No regex in `std`. Either a generated scanner (the normalized pattern is a DFA-shaped + tree already) or a dependency. **This is the second real friction**, and it is the one that + would decide whether the Rust backend needs a generated `re` module. +- **Strings.** `str.len` on a non-ASCII value is `s.chars().count()`; scalar comparison is native + (`Ord` on `str` compares bytes, which is code point order), so Rust behaves like Go here. + +## What has no clean lowering + +1. **The default capability environment**, because HTTP is not in `std`. +2. **Regex**, for the same reason; a generated scanner is the principled answer and is more work + than the other twelve intrinsics put together. +3. **Arbitrary-precision integers**, if a range ever demands them: `i128` is the ceiling without a + dependency. + +Everything else maps directly, and two of the three frictions are the same friction — `std` is +smaller than the other three targets' standard libraries. The architecture is not falsified by +Rust; the "no dependencies" rule is the thing that would have to bend, and it would bend for +exactly two operations. diff --git a/engine/docs/targets/rust.md b/engine/docs/targets/rust.md new file mode 100644 index 000000000..a52de5258 --- /dev/null +++ b/engine/docs/targets/rust.md @@ -0,0 +1,157 @@ +# Target: Rust + +**Baseline** Rust 2021, `std` only. **Formatter** `rustfmt` (run as `cargo fmt`, which is rustfmt +applied to every file a `Cargo.toml` lists). **Linters** `cargo clippy -- -D warnings`. + +## Representation + +| Semantic type | Rust | +|---|---| +| `Int[lo..hi]`, `Decimal`, `CivilDate`, `Instant`, `Duration` | `i64` | +| `Float` | `f64` | +| `String`, `Ascii`, `Digits`, `Enum` | `String` | +| `List` | `Vec` | +| `Option` | `Option` | +| `Record` | `#[derive(Clone, Debug, PartialEq)] struct` with `pub` fields | +| Capability environment | `&dyn Capabilities` | + +Every **local and struct field** is owned wherever it is bound, never `&str`/`&[T]` — a `let`, a +record field, a list item, `Some(...)`, `return`, always builds an owned value. A **function +parameter** of `String`/`Enum`/`List` type borrows (`&str`/`&[T]`) instead when a whole-program +pre-pass proves it is never returned, stored, assigned to or forwarded to another owned parameter; +otherwise it stays owned, the same as everything else. `docs/targets/rust-sketch.md` planned +parameters borrowing from the start, on a single-function argument that did not survive contact +with the shared Target AST; [ADR 0009](../decisions/0009-rust-values-are-owned.md) records exactly +why not, and [ADR 0010](../decisions/0010-rust-parameters-borrow-where-sound.md) records what +changed to make the sketch's original claim about parameters true after all — a pre-pass over the +*whole* `CProgram`, computed once before lowering starts, rather than a decision made function by +function during lowering or printing. `analysis/borrows.ts` is the pre-pass; `TParam.borrowed` and +a "call" node's `borrowedArgs` (`backend/tast.ts`) are what it hands the lowerer and the printer. + +## Shape of the output + +- One crate, `coreout`: `src/lib.rs` declares `pub mod` for every generated module plus `support` + and (when the project declares any) `errors`, and re-exports each flatly (`pub use lib_digits::*;` + and so on). A module calls another's function unqualified — Go's advantage of a single package, + recovered here through the glob rather than through the language having no module system at all. +- A utility is `pub` (reachable at the crate root through the flat re-export above). A helper its + own source module exports but that is not itself a utility — reachable across generated modules, + never meant to be reachable from outside the crate — is `pub(crate)`: visible to the `use + crate::*;` every module imports, but a glob re-export silently drops it rather than leaking it + further (verified against `rustc` directly: a `pub(crate)` item never surfaces through `pub use + module::*;`). A helper never exported at all, called only from within its own module, is a plain + `fn` — Rust's own notion of private, and the tightest of the three. +- `support.rs` holds the `Capabilities` trait and its request/response records, the generic + sequence and string helpers the capability table names, and the regex matcher (below). +- **No default `Capabilities`.** `support.rs` declares the trait only — no HTTP client, no clock, + no RNG anywhere in the `coreout` crate, on purpose: the library depends on `std` alone. The + differential driver's `src/bin/driver.rs` builds its own fixture-backed `FakeCapabilities`, but + that fake lives in the driver binary, never in the library crate. This means a utility whose + effects reach `Http`, `Clock` or `Random` (`get_address_info_by_cep`, `generate_cpf`, + `generate_cnpj`) cannot get the public-wrapper treatment TypeScript and Python give the same + utilities (`docs/semantics.md` §4.1) — there is no default to hand a wrapper, and one is not + fabricated to manufacture the appearance of parity. The capability-taking function stays the + only entry point, under its original name, taking `&dyn Capabilities` as a normal parameter a + caller supplies. This is a genuine, reported gap from drop-in replacement, not a defect in this + backend's generation — see [ADR 0011](../decisions/0011-public-entry-points-vs-capabilities.md). +- `errors.rs` holds one flat `CoreError` enum, one variant per declared domain error — not one type + per utility, which is the other half of ADR 0009. +- `Fail` is `Result`. The shared lowerer's Go-shaped hoist + (`let (v, err) = f(); if err != nil { return zero, err }`) collapses back into `let v = f()?;` at + print time, which reads as idiomatic Rust specifically because every fallible function shares one + error type, so `?` never has to bridge a mismatch. +- Rust has a conditional expression, so `cond` prints as `if … { … } else { … }` directly — no + statement hoist, unlike Go. +- `seq.fold`, `seq.map` and `seq.filter` are loops, matching Go; `seq.any`, `seq.all`, `seq.find` + and `seq.sortStableBy` stay as intrinsic calls taking a closure. +- `task.race` is `std::thread::scope` plus an `mpsc` channel: one thread per task, the first `Some` + received wins, exactly Go's `raceFirstSome` shape. The closures borrow rather than `move`, which + is what lets the same argument be handed to every task (see ADR 0009). +- The differential driver (`src/bin/driver.rs`) hand-writes a minimal JSON reader/writer + (`src/json.rs`) for its own `{"fn":…,"args":[…]}` protocol: `std` has no JSON, and this is driver + code, not project code, in the same sense the regex matcher below is. + +## Notable lowerings + +**`str.len`** on a proven-ASCII value is `s.len()` (bytes, cheap); otherwise it is +`s.chars().count()`, which walks scalars without allocating — unlike Go's `[]rune(s)` conversion, +which the coordinator's own benchmark found recompiling a *pattern* from source on every call, not +counting scalars, but the same "does this walk allocate" question applies here too, and here the +answer is no. + +**`str.compare`** is native with no precondition: `Ord` on `str` compares UTF-8 bytes, which is +code point order — like Go, unlike JavaScript's UTF-16 comparison. This is the sketch's headline +claim and it holds exactly as predicted. + +**`re.test`**. `std` has no regex engine — the sketch's second predicted friction, confirmed. The +first fix (landed with this target) was to stop recompiling anything at call time: each project +pattern became a `pub static` `ReNode` tree, a `const` expression rustc places in the binary's +read-only data once. That was necessary but not sufficient — a later cross-language benchmark +(200 000 iterations of `isValidCpf`/`isValidCnpj` against the handwritten `brazilian-utils/rust` +crate) found the generated code 24-50x slower even with compilation gone, because the *matcher* +itself, walking that `ReNode` tree, allocated a fresh `Vec` of reachable positions per node +per position and sorted and deduplicated each one: dozens of heap allocations per call for a +14-character input. `support::re_test` alone was 79% of the call. + +The engine knows every pattern in a project before it generates a line of Rust, so the real fix is +to stop interpreting a tree at call time at all. `engine/src/targets/rust/index.ts`'s "Regex" +section compiles each pattern into one of two things, decided once, at generation time: + +- **A dedicated scanner** (`re_match_N`, in `support.rs`), for a pattern that is a top-level chain + of character classes and repeated character classes with no alternation and no repeated group — + every pattern `core/source` actually uses. `chainElementsOf` additionally requires that any + variable-length run have a class disjoint from whatever immediately follows it (maximal munch: + the two classes can never disagree about where one run ends and the next begins), which is what + lets the emitted scanner consume each run greedily, in one forward pass over `&str`, and never + need to back off. It is built from two small generic helpers, `re_take_fixed`/`re_take_class`, + that slice the input forward; nothing here allocates. +- **The fallback matcher** (`re_test`/`ReNode`, still in `support.rs`), for anything + `chainElementsOf` refuses — alternation anywhere, or a repeated group more complex than one + class. This is no longer the position-set NFA: it is continuation-passing backtracking over + `&str` byte slices, a direct port of `regex.ts`'s own reference matcher (`matchNode`), which is + what every accepted pattern is already checked against in `engine/tests/regex.spec.ts`. Its + continuations are stack-local closures borrowed with `&dyn Fn`, never boxed, so it does not + allocate either — the `Vec` `re_test` used to collect the input into, and the per-node + `Vec`, are both just gone, not replaced with a differently-shaped allocation. + +No pattern in `core/source` takes the fallback path today; `LOWERING.md`'s `re.test` row and the +`because` string in `RUST_CANDIDATES` name the rule that decides, so a reviewer does not have to +infer it from which patterns happen to appear. Re-measured, `support::re_match_3` (the CPF +pattern's scanner) alone is on the order of the input's own length to walk once — see +`core/bench/README.md`'s Rust section for the current numbers and what dominates the call now that +the matcher no longer does. + +**Parameter borrowing.** `analysis/borrows.ts` decides, once per build and before lowering starts, +which `String`/`Enum`/`List` parameters may print as `&str`/`&[T]`; [ADR +0010](../decisions/0010-rust-parameters-borrow-where-sound.md) has the full rule set and why +starting optimistic and demoting on evidence is sound. In practice this reaches deepest along a +read-only chain: `digit_at(value: &str, index: i64)` only reads a byte, `cpf_check_digit(cpf: &str, +size: i64)` only forwards that byte read in a loop, and `is_valid_cpf(cpf: &str)` only forwards +`cpf` into `keep_digits` and the trim — none of the three needs to own the 11-digit string, so +none of them do, and the loop that used to clone it once per weight (`digit_at(cpf.to_owned(), +index)`, 9 to 11 times per call) now passes a bare `&str` copy instead. `cnpj_check_digit(cnpj: +&str, weights: &[i64])` borrows both parameters the same way, including the hoisted weight table +(`LIB_CNPJ_TABLE1: &[i64]`, already a reference — passed bare, not `&`-wrapped again). A record +parameter (`FormatCnpjOptions`, `AddressInfo`) and every return type stay owned regardless; ADR +0010 explains why a borrowed struct field is a different, larger change this decision does not +make. + +**`int.max`/`int.min`** detect a nested clamp (`x.max(lo).min(hi)`, the shape the source's own +`int.min(int.max(x, lo), hi)` prints as by default) and merge it into `.clamp(lo, hi)`, which +`clippy::manual_clamp` (default warn) asks for. This is a printer-level rewrite, not a new +intrinsic: the Core has no clamp primitive, and adding one for one target would be the wrong fix. + +**`x >= lo && x <= hi`** (and the `||`-negated shape `x < lo || x > hi`) prints as +`(lo..=hi).contains(&x)` / `!(lo..=hi).contains(&x)`. The Core has no range type (`docs/semantics.md` +admits `<`/`<=`/`>`/`>=`, not a bounds check), so a source author always writes the pair, and +`clippy::manual_range_contains` (default warn) always asks for the idiom instead. Recognized in the +printer's `binary` case, which is where every occurrence — however deeply nested a source +expression buries it — is reached exactly once, recursively, regardless of where it came from. + +**`opt.unwrap`**, printed as a bare `.unwrap()`, moves the `Option` it unwraps. A narrowed read — +`if (response === undefined) return …` followed by two or three further reads of `response.field` +— re-runs `opt.unwrap` at *every* read (the Core never re-types a local narrower; see ADR 0009's +Go comparison), so a plain `.unwrap()` would move `response` out on the first read and leave the +second one looking at a moved value. A read reached through `.unwrap()` goes through `.as_ref()` +first instead, a borrow, which is what makes any number of narrowed reads work the way a Go pointer +dereference already does for free. diff --git a/engine/docs/targets/typescript.md b/engine/docs/targets/typescript.md new file mode 100644 index 000000000..c5c9c143a --- /dev/null +++ b/engine/docs/targets/typescript.md @@ -0,0 +1,63 @@ +# Target: TypeScript + +**Baseline** ES2020 on Node 20. **Dependencies** none. **Formatter** Prettier. +**Linters** `tsc --strict --noEmit`. + +## Representation + +| Semantic type | TypeScript | +|---|---| +| `Int[lo..hi]` | `number` while the range fits ±(2^53 − 1), `bigint` otherwise | +| `Float` | `number` | +| `Decimal` | `number` holding the unscaled integer | +| `String`, `Ascii`, `Digits` | `string` | +| `List` | `readonly T[]`, and `T[]` while a local is still being built | +| `Option` | `T \| undefined` | +| `Record` | an exported `type` with `readonly` fields | +| `Enum` | a union of string literals | +| `CivilDate`, `Instant`, `Duration` | `number` | + +## Shape of the output + +- ESM only, one module per source module, named exports, no default export. +- A module exports exactly what its source module exports, no more: a function the source never + wrote `export` on (`isOk` alongside `getAddressInfoByCep`, say) prints as a plain, unexported + `function`, callable from elsewhere in the same file but invisible to an importer — TypeScript's + own notion of a module-private helper, the one every other file in this package already uses. +- No effect at module load: constant data is inlined as literals, and nothing constructs a `Map`, + a `Set` or a `RegExp` at the top level (`capabilities.ts`'s `DEFAULT_CAPABILITIES`, below, is the + one deliberate exception). +- Relative imports carry their extension, so Node runs the generated sources directly. +- A function that reaches `Http` is `async`, and its callers await it. That colouring is computed + from the effect set, never written by an author. +- Domain errors are classes in `errors.ts`, extending a generated `DomainError`. +- `capabilities.ts` holds the generated default environment (`fetch`, timers, `Math.random`) and + the `raceFirstSome` helper. It is generated code, not a package: a test passes a different + object. `DEFAULT_CAPABILITIES` is that environment built once, at module load, not per call — + every public wrapper (below) shares the one instance, the way a caller who builds their own + environment would share it across calls rather than rebuild it. +- **A utility that reaches `Http`, `Clock` or `Random` is a public wrapper over an internal seam** + (`docs/semantics.md` §4.1, [ADR 0011](../decisions/0011-public-entry-points-vs-capabilities.md)): + `getAddressInfoByCep(cep)` keeps the source's exact signature and calls + `getAddressInfoByCepWith(cep, DEFAULT_CAPABILITIES)`, an exported sibling in the same module the + differential driver calls directly to inject a fixture-backed fake. Both are `export`ed — + the wrapper because it is the utility, the seam because the driver has to reach it from + `_driver.ts` — but only the wrapper is listed as a utility in `API.json`; the seam is marked + there under `seams` instead. + +## Notable lowerings + +`String#length` counts UTF-16 code units, so it is selected only for proven-ASCII values; +otherwise the length is `[...value].length`. `charCodeAt`, `slice`, `indexOf` and `padStart` need +the same proof, for the same reason. + +`toUpperCase` is selected only for proven-ASCII values: it maps "ß" to "SS", which ASCII-only case +mapping does not. + +`<` on strings compares UTF-16 code units, which orders astral scalars below U+E000. The native +comparison is therefore selected only for proven-ASCII operands, and everything else uses +`std/strings::compareScalars`. This is the clearest case in the system of a refinement paying for +itself. + +`Array#sort` has been required to be stable since ES2019, so `seq.sortStable` is a native sort on +a copy. diff --git a/engine/examples/generic/engine.config.json b/engine/examples/generic/engine.config.json new file mode 100644 index 000000000..7eaea4520 --- /dev/null +++ b/engine/examples/generic/engine.config.json @@ -0,0 +1,6 @@ +{ + "name": "generic-example", + "sourceRoot": "source", + "out": "out", + "targets": ["typescript", "python", "go"] +} diff --git a/engine/examples/generic/source/is-valid-luhn.ts b/engine/examples/generic/source/is-valid-luhn.ts new file mode 100644 index 000000000..fd77315f8 --- /dev/null +++ b/engine/examples/generic/source/is-valid-luhn.ts @@ -0,0 +1,28 @@ +/** + * A deliberately domain-neutral example: the engine knows nothing about any particular library. + * + * Luhn is the check digit rule behind credit card numbers, IMEIs and several national identifiers + * (ISO/IEC 7812-1, annex B). + */ + +const DIGITS = /^[0-9]{2,19}$/; + +/** Whether a digit string satisfies the Luhn check. */ +export function isValidLuhn(value: string): boolean { + if (!re.test(DIGITS, value)) { + return false; + } + + let sum: IntRange<0, 200> = 0; + const scalars = str.codePoints(value); + + for (let index = 0; index < scalars.length; index++) { + const digit = (seq.at(scalars, index) ?? 48) - 48; + const doubled = (scalars.length - index) % 2 === 0; + const weighted = doubled ? digit * 2 : digit; + + sum += weighted > 9 ? weighted - 9 : weighted; + } + + return sum % 10 === 0; +} diff --git a/engine/examples/generic/source/slugify.ts b/engine/examples/generic/source/slugify.ts new file mode 100644 index 000000000..eb8f529ed --- /dev/null +++ b/engine/examples/generic/source/slugify.ts @@ -0,0 +1,32 @@ +/** + * A second example: turning a title into an ASCII slug, with no host locale involved. + */ + +const HYPHEN = 45; +const LOWER_A = 97; +const LOWER_Z = 122; +const ZERO = 48; +const NINE = 57; + +/** A lower cased, hyphen separated slug, keeping only ASCII letters and digits. */ +export function slugify(title: string): Ascii { + let out: IntRange<0, 127>[] = []; + let pendingHyphen = false; + + for (const point of str.codePoints(str.asciiLower(title))) { + // The guard is written inline rather than behind a predicate because the checker reads a + // guard, not a called function: this is what proves every scalar pushed is ASCII. + if ((point >= LOWER_A && point <= LOWER_Z) || (point >= ZERO && point <= NINE)) { + if (pendingHyphen && out.length > 0) { + out.push(HYPHEN); + } + + pendingHyphen = false; + out.push(point); + } else { + pendingHyphen = true; + } + } + + return str.fromCodePoints(out); +} diff --git a/engine/package-lock.json b/engine/package-lock.json new file mode 100644 index 000000000..175e341ed --- /dev/null +++ b/engine/package-lock.json @@ -0,0 +1,908 @@ +{ + "name": "@logic-engine/compiler", + "version": "0.1.0", + "lockfileVersion": 3, + "requires": true, + "packages": { + "": { + "name": "@logic-engine/compiler", + "version": "0.1.0", + "license": "MIT", + "dependencies": { + "oxc-parser": "0.150.0" + }, + "bin": { + "logic-engine": "src/cli.ts" + }, + "devDependencies": { + "@types/node": "24.13.6", + "esbuild": "0.27.4", + "prettier": "3.8.1", + "typescript": "5.9.3" + }, + "engines": { + "node": ">=22.12.0" + } + }, + "node_modules/@esbuild/aix-ppc64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/aix-ppc64/-/aix-ppc64-0.27.4.tgz", + "integrity": "sha512-cQPwL2mp2nSmHHJlCyoXgHGhbEPMrEEU5xhkcy3Hs/O7nGZqEpZ2sUtLaL9MORLtDfRvVl2/3PAuEkYZH0Ty8Q==", + "cpu": [ + "ppc64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "aix" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/android-arm": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/android-arm/-/android-arm-0.27.4.tgz", + "integrity": "sha512-X9bUgvxiC8CHAGKYufLIHGXPJWnr0OCdR0anD2e21vdvgCI8lIfqFbnoeOz7lBjdrAGUhqLZLcQo6MLhTO2DKQ==", + "cpu": [ + "arm" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/android-arm64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/android-arm64/-/android-arm64-0.27.4.tgz", + "integrity": "sha512-gdLscB7v75wRfu7QSm/zg6Rx29VLdy9eTr2t44sfTW7CxwAtQghZ4ZnqHk3/ogz7xao0QAgrkradbBzcqFPasw==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/android-x64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/android-x64/-/android-x64-0.27.4.tgz", + "integrity": "sha512-PzPFnBNVF292sfpfhiyiXCGSn9HZg5BcAz+ivBuSsl6Rk4ga1oEXAamhOXRFyMcjwr2DVtm40G65N3GLeH1Lvw==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/darwin-arm64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/darwin-arm64/-/darwin-arm64-0.27.4.tgz", + "integrity": "sha512-b7xaGIwdJlht8ZFCvMkpDN6uiSmnxxK56N2GDTMYPr2/gzvfdQN8rTfBsvVKmIVY/X7EM+/hJKEIbbHs9oA4tQ==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/darwin-x64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/darwin-x64/-/darwin-x64-0.27.4.tgz", + "integrity": "sha512-sR+OiKLwd15nmCdqpXMnuJ9W2kpy0KigzqScqHI3Hqwr7IXxBp3Yva+yJwoqh7rE8V77tdoheRYataNKL4QrPw==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/freebsd-arm64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/freebsd-arm64/-/freebsd-arm64-0.27.4.tgz", + "integrity": "sha512-jnfpKe+p79tCnm4GVav68A7tUFeKQwQyLgESwEAUzyxk/TJr4QdGog9sqWNcUbr/bZt/O/HXouspuQDd9JxFSw==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "freebsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/freebsd-x64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/freebsd-x64/-/freebsd-x64-0.27.4.tgz", + "integrity": "sha512-2kb4ceA/CpfUrIcTUl1wrP/9ad9Atrp5J94Lq69w7UwOMolPIGrfLSvAKJp0RTvkPPyn6CIWrNy13kyLikZRZQ==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "freebsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-arm": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/linux-arm/-/linux-arm-0.27.4.tgz", + "integrity": "sha512-aBYgcIxX/wd5n2ys0yESGeYMGF+pv6g0DhZr3G1ZG4jMfruU9Tl1i2Z+Wnj9/KjGz1lTLCcorqE2viePZqj4Eg==", + "cpu": [ + "arm" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-arm64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/linux-arm64/-/linux-arm64-0.27.4.tgz", + "integrity": "sha512-7nQOttdzVGth1iz57kxg9uCz57dxQLHWxopL6mYuYthohPKEK0vU0C3O21CcBK6KDlkYVcnDXY099HcCDXd9dA==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-ia32": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/linux-ia32/-/linux-ia32-0.27.4.tgz", + "integrity": "sha512-oPtixtAIzgvzYcKBQM/qZ3R+9TEUd1aNJQu0HhGyqtx6oS7qTpvjheIWBbes4+qu1bNlo2V4cbkISr8q6gRBFA==", + "cpu": [ + "ia32" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-loong64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/linux-loong64/-/linux-loong64-0.27.4.tgz", + "integrity": "sha512-8mL/vh8qeCoRcFH2nM8wm5uJP+ZcVYGGayMavi8GmRJjuI3g1v6Z7Ni0JJKAJW+m0EtUuARb6Lmp4hMjzCBWzA==", + "cpu": [ + "loong64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-mips64el": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/linux-mips64el/-/linux-mips64el-0.27.4.tgz", + "integrity": "sha512-1RdrWFFiiLIW7LQq9Q2NES+HiD4NyT8Itj9AUeCl0IVCA459WnPhREKgwrpaIfTOe+/2rdntisegiPWn/r/aAw==", + "cpu": [ + "mips64el" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-ppc64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/linux-ppc64/-/linux-ppc64-0.27.4.tgz", + "integrity": "sha512-tLCwNG47l3sd9lpfyx9LAGEGItCUeRCWeAx6x2Jmbav65nAwoPXfewtAdtbtit/pJFLUWOhpv0FpS6GQAmPrHA==", + "cpu": [ + "ppc64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-riscv64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/linux-riscv64/-/linux-riscv64-0.27.4.tgz", + "integrity": "sha512-BnASypppbUWyqjd1KIpU4AUBiIhVr6YlHx/cnPgqEkNoVOhHg+YiSVxM1RLfiy4t9cAulbRGTNCKOcqHrEQLIw==", + "cpu": [ + "riscv64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-s390x": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/linux-s390x/-/linux-s390x-0.27.4.tgz", + "integrity": "sha512-+eUqgb/Z7vxVLezG8bVB9SfBie89gMueS+I0xYh2tJdw3vqA/0ImZJ2ROeWwVJN59ihBeZ7Tu92dF/5dy5FttA==", + "cpu": [ + "s390x" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/linux-x64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/linux-x64/-/linux-x64-0.27.4.tgz", + "integrity": "sha512-S5qOXrKV8BQEzJPVxAwnryi2+Iq5pB40gTEIT69BQONqR7JH1EPIcQ/Uiv9mCnn05jff9umq/5nqzxlqTOg9NA==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/netbsd-arm64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/netbsd-arm64/-/netbsd-arm64-0.27.4.tgz", + "integrity": "sha512-xHT8X4sb0GS8qTqiwzHqpY00C95DPAq7nAwX35Ie/s+LO9830hrMd3oX0ZMKLvy7vsonee73x0lmcdOVXFzd6Q==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "netbsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/netbsd-x64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/netbsd-x64/-/netbsd-x64-0.27.4.tgz", + "integrity": "sha512-RugOvOdXfdyi5Tyv40kgQnI0byv66BFgAqjdgtAKqHoZTbTF2QqfQrFwa7cHEORJf6X2ht+l9ABLMP0dnKYsgg==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "netbsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/openbsd-arm64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/openbsd-arm64/-/openbsd-arm64-0.27.4.tgz", + "integrity": "sha512-2MyL3IAaTX+1/qP0O1SwskwcwCoOI4kV2IBX1xYnDDqthmq5ArrW94qSIKCAuRraMgPOmG0RDTA74mzYNQA9ow==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "openbsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/openbsd-x64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/openbsd-x64/-/openbsd-x64-0.27.4.tgz", + "integrity": "sha512-u8fg/jQ5aQDfsnIV6+KwLOf1CmJnfu1ShpwqdwC0uA7ZPwFws55Ngc12vBdeUdnuWoQYx/SOQLGDcdlfXhYmXQ==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "openbsd" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/openharmony-arm64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/openharmony-arm64/-/openharmony-arm64-0.27.4.tgz", + "integrity": "sha512-JkTZrl6VbyO8lDQO3yv26nNr2RM2yZzNrNHEsj9bm6dOwwu9OYN28CjzZkH57bh4w0I2F7IodpQvUAEd1mbWXg==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "openharmony" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/sunos-x64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/sunos-x64/-/sunos-x64-0.27.4.tgz", + "integrity": "sha512-/gOzgaewZJfeJTlsWhvUEmUG4tWEY2Spp5M20INYRg2ZKl9QPO3QEEgPeRtLjEWSW8FilRNacPOg8R1uaYkA6g==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "sunos" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/win32-arm64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/win32-arm64/-/win32-arm64-0.27.4.tgz", + "integrity": "sha512-Z9SExBg2y32smoDQdf1HRwHRt6vAHLXcxD2uGgO/v2jK7Y718Ix4ndsbNMU/+1Qiem9OiOdaqitioZwxivhXYg==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/win32-ia32": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/win32-ia32/-/win32-ia32-0.27.4.tgz", + "integrity": "sha512-DAyGLS0Jz5G5iixEbMHi5KdiApqHBWMGzTtMiJ72ZOLhbu/bzxgAe8Ue8CTS3n3HbIUHQz/L51yMdGMeoxXNJw==", + "cpu": [ + "ia32" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@esbuild/win32-x64": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/@esbuild/win32-x64/-/win32-x64-0.27.4.tgz", + "integrity": "sha512-+knoa0BDoeXgkNvvV1vvbZX4+hizelrkwmGJBdT17t8FNPwG2lKemmuMZlmaNQ3ws3DKKCxpb4zRZEIp3UxFCg==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">=18" + } + }, + "node_modules/@oxc-parser/binding-android-arm-eabi": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-android-arm-eabi/-/binding-android-arm-eabi-0.150.0.tgz", + "integrity": "sha512-oQef2Zu4Prz1KLKznz3HqZzU9uVoA5PMoDZuuLmqms7hKmKSAPzlaMnLllJq3t+rgKmfJJ2siPrpZfFrW06btw==", + "cpu": [ + "arm" + ], + "license": "MIT", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-android-arm64": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-android-arm64/-/binding-android-arm64-0.150.0.tgz", + "integrity": "sha512-B6ofpoFiAUwIZ0MJ2IgHPvZK8FAtL4qzSZRwOkAKIfxEobE1mQC8nDdiFZj7pUaJiVUVCsfkMsBJKedzO8rZvA==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-darwin-arm64": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-darwin-arm64/-/binding-darwin-arm64-0.150.0.tgz", + "integrity": "sha512-J+9IHKzx/bSz1JetOfD4zKXSK9sOm4/a7+0qomJODcTL6JsRXGeV/ZdkAPSIDydfFmWzYCraMlQEosSZEyBYkQ==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-darwin-x64": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-darwin-x64/-/binding-darwin-x64-0.150.0.tgz", + "integrity": "sha512-v6IPfcAcSYrBWXV5Tce1DmxDMXLhEYpIIWiRFPJopgCVscZdLaV4MRQOUxvL3usQlw+yrwYOjBFB9IwBx+jUiQ==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-freebsd-x64": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-freebsd-x64/-/binding-freebsd-x64-0.150.0.tgz", + "integrity": "sha512-AoR/4jD02HET0KO0yNT4nbTE+XJUxiO9jB1X/ycEjUE3WrIJ3IjV/Vf+gskhWLcqqeXfvNnkDtBR7epllAm88A==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "freebsd" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-linux-arm-gnueabihf": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-linux-arm-gnueabihf/-/binding-linux-arm-gnueabihf-0.150.0.tgz", + "integrity": "sha512-A/hycCFjLUrCmLoL5O/vBp40ohlaXO1Ta3v5hqYZxicYFs129wZmgZJggcW4mYGYbZebQLhup+N8JjeQl8qwyw==", + "cpu": [ + "arm" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-linux-arm-musleabihf": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-linux-arm-musleabihf/-/binding-linux-arm-musleabihf-0.150.0.tgz", + "integrity": "sha512-pbqahg1Pkz7J4RKmXm30s/iQu99hv64ayXCu8P4p75HcEH5azOdR4eGqiB6lFGkFR3mMpzaYsSKLkS8oJbjmeQ==", + "cpu": [ + "arm" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-linux-arm64-gnu": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-linux-arm64-gnu/-/binding-linux-arm64-gnu-0.150.0.tgz", + "integrity": "sha512-HV11aRbBQGwqv8bEo+K/6qr88uB4fEe1mhXncMhlVCrj+WpBSraxrUzHaPVmeQRU6x6x8v/3DuCbbo93rpp+fw==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-linux-arm64-musl": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-linux-arm64-musl/-/binding-linux-arm64-musl-0.150.0.tgz", + "integrity": "sha512-k6pVkJqALwtEuP4zukVhGRhdIy4+ofChUIUUsAHHyqDZlNsknmel3JjbggrJiWGaes6h/dIcWmEXsKW4SmK9oQ==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-linux-ppc64-gnu": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-linux-ppc64-gnu/-/binding-linux-ppc64-gnu-0.150.0.tgz", + "integrity": "sha512-tEFg39mw/rHO5n8GK2DR4ZMFfsPQZtnQIPbsIpjoQRomJc/2tRfsCSmCT6gNit+NZGoeJG5aH9ATDH5bf0WnoQ==", + "cpu": [ + "ppc64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-linux-riscv64-gnu": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-linux-riscv64-gnu/-/binding-linux-riscv64-gnu-0.150.0.tgz", + "integrity": "sha512-0JO7IFkoek6HqV589Menx3hJySNXi6quRhO4tq2P7kpEHKEXoDATnoPVUpn5B7gDH2KLkdo751N0uIrGJFCTCw==", + "cpu": [ + "riscv64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-linux-riscv64-musl": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-linux-riscv64-musl/-/binding-linux-riscv64-musl-0.150.0.tgz", + "integrity": "sha512-7XPzREnyAS5wHU9aB+49Uaqgoo1gVCklHc3DRlc3fJy2tXD/eAJFN1QFz/crj+Ulup5P1+axNREF+/RGaB/IRA==", + "cpu": [ + "riscv64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-linux-s390x-gnu": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-linux-s390x-gnu/-/binding-linux-s390x-gnu-0.150.0.tgz", + "integrity": "sha512-7n7ZxcbRWDFaTdyj8M8p0O9OfzoMKm8O6syBAIgcSutbOALq0Bb9G52Me+XH0anMLZV+NgXc726gqgG2q39vcQ==", + "cpu": [ + "s390x" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-linux-x64-gnu": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-linux-x64-gnu/-/binding-linux-x64-gnu-0.150.0.tgz", + "integrity": "sha512-Vx0GSA9ZCRTUiczoEQnIIyXkMvtEY8uc6IiwLQtMe5ci2sXPqjrry0Ek5fsPP1k2kCzkML3ow9UOOEgnEDi7Bw==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-linux-x64-musl": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-linux-x64-musl/-/binding-linux-x64-musl-0.150.0.tgz", + "integrity": "sha512-ii4/9m3viDLssMnfXU7/Pni/3nYplApuGayceX4qsVGaqqQJCx3tHIH9M/HWIeSty5i8LjS21zPYT6lVIkwLxw==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-openharmony-arm64": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-openharmony-arm64/-/binding-openharmony-arm64-0.150.0.tgz", + "integrity": "sha512-ZtoxX5rez36ZWkwtiEhjkCPnlTPoQ2I7lIL0IYZ1yjxMYohbvoOjBmUIZzrf9W/HouONq0omIfTeCWEY+R380w==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "openharmony" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-win32-arm64-msvc": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-win32-arm64-msvc/-/binding-win32-arm64-msvc-0.150.0.tgz", + "integrity": "sha512-VqeRb5JX/bKYBrff7TUDvJyKY0914UzuCkuXIGxJS4FUmvMNfqkCbc7Sq90iMz9DlmAtlggwaBYQ9eg3h3ltiw==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-win32-ia32-msvc": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-win32-ia32-msvc/-/binding-win32-ia32-msvc-0.150.0.tgz", + "integrity": "sha512-EPqJfeZ4Pgg2BJSsCV8GozxV0FDltRzpVtGVa1r2noJu1iu+opOw3gtBWXwIo9odCT/MdFfyHxdy0/CRqs3OqA==", + "cpu": [ + "ia32" + ], + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-parser/binding-win32-x64-msvc": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-parser/binding-win32-x64-msvc/-/binding-win32-x64-msvc-0.150.0.tgz", + "integrity": "sha512-n5YMzbqwPQqozbkrexi0lNZrUz5cnPHalYpFWe6Dau7AmKIWNo+y24IwzDuZEQtEdUHuKhXmMMih/MPjOI+CMw==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": "^20.19.0 || >=22.12.0" + } + }, + "node_modules/@oxc-project/types": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/@oxc-project/types/-/types-0.150.0.tgz", + "integrity": "sha512-rDS5/31E9HfPl/CIzGrn0DOlvBbXFseQ5URJ9sYMfstbKLD/c6Gm9vmRzRGDdAXyOIL4zmO37lc9RIwYqVruZw==", + "license": "MIT", + "funding": { + "url": "https://github.com/sponsors/oxc-project" + } + }, + "node_modules/@types/node": { + "version": "24.13.6", + "resolved": "https://registry.npmjs.org/@types/node/-/node-24.13.6.tgz", + "integrity": "sha512-SGrw/h3KPFshy3OE6ZL53LMBG5vGQQ8/gIpiqz/kRZhPJ7HgwCEs8LBuNtWLa8dvGZVpSF7+Bf+c11HUrCb/yg==", + "dev": true, + "license": "MIT", + "dependencies": { + "undici-types": "~7.18.0" + } + }, + "node_modules/esbuild": { + "version": "0.27.4", + "resolved": "https://registry.npmjs.org/esbuild/-/esbuild-0.27.4.tgz", + "integrity": "sha512-Rq4vbHnYkK5fws5NF7MYTU68FPRE1ajX7heQ/8QXXWqNgqqJ/GkmmyxIzUnf2Sr/bakf8l54716CcMGHYhMrrQ==", + "dev": true, + "hasInstallScript": true, + "license": "MIT", + "bin": { + "esbuild": "bin/esbuild" + }, + "engines": { + "node": ">=18" + }, + "optionalDependencies": { + "@esbuild/aix-ppc64": "0.27.4", + "@esbuild/android-arm": "0.27.4", + "@esbuild/android-arm64": "0.27.4", + "@esbuild/android-x64": "0.27.4", + "@esbuild/darwin-arm64": "0.27.4", + "@esbuild/darwin-x64": "0.27.4", + "@esbuild/freebsd-arm64": "0.27.4", + "@esbuild/freebsd-x64": "0.27.4", + "@esbuild/linux-arm": "0.27.4", + "@esbuild/linux-arm64": "0.27.4", + "@esbuild/linux-ia32": "0.27.4", + "@esbuild/linux-loong64": "0.27.4", + "@esbuild/linux-mips64el": "0.27.4", + "@esbuild/linux-ppc64": "0.27.4", + "@esbuild/linux-riscv64": "0.27.4", + "@esbuild/linux-s390x": "0.27.4", + "@esbuild/linux-x64": "0.27.4", + "@esbuild/netbsd-arm64": "0.27.4", + "@esbuild/netbsd-x64": "0.27.4", + "@esbuild/openbsd-arm64": "0.27.4", + "@esbuild/openbsd-x64": "0.27.4", + "@esbuild/openharmony-arm64": "0.27.4", + "@esbuild/sunos-x64": "0.27.4", + "@esbuild/win32-arm64": "0.27.4", + "@esbuild/win32-ia32": "0.27.4", + "@esbuild/win32-x64": "0.27.4" + } + }, + "node_modules/oxc-parser": { + "version": "0.150.0", + "resolved": "https://registry.npmjs.org/oxc-parser/-/oxc-parser-0.150.0.tgz", + "integrity": "sha512-zwajKw1GUa57IOdafIS7/yOObGF2NihwIfZfs5yJTRprMSkiwhWOc/k+n8bwnUPbHr2/XsjfBO8KyTPaW3VpWg==", + "license": "MIT", + "dependencies": { + "@oxc-project/types": "^0.150.0" + }, + "engines": { + "node": "^20.19.0 || >=22.12.0" + }, + "funding": { + "url": "https://github.com/sponsors/oxc-project" + }, + "optionalDependencies": { + "@oxc-parser/binding-android-arm-eabi": "0.150.0", + "@oxc-parser/binding-android-arm64": "0.150.0", + "@oxc-parser/binding-darwin-arm64": "0.150.0", + "@oxc-parser/binding-darwin-x64": "0.150.0", + "@oxc-parser/binding-freebsd-x64": "0.150.0", + "@oxc-parser/binding-linux-arm-gnueabihf": "0.150.0", + "@oxc-parser/binding-linux-arm-musleabihf": "0.150.0", + "@oxc-parser/binding-linux-arm64-gnu": "0.150.0", + "@oxc-parser/binding-linux-arm64-musl": "0.150.0", + "@oxc-parser/binding-linux-ppc64-gnu": "0.150.0", + "@oxc-parser/binding-linux-riscv64-gnu": "0.150.0", + "@oxc-parser/binding-linux-riscv64-musl": "0.150.0", + "@oxc-parser/binding-linux-s390x-gnu": "0.150.0", + "@oxc-parser/binding-linux-x64-gnu": "0.150.0", + "@oxc-parser/binding-linux-x64-musl": "0.150.0", + "@oxc-parser/binding-openharmony-arm64": "0.150.0", + "@oxc-parser/binding-win32-arm64-msvc": "0.150.0", + "@oxc-parser/binding-win32-ia32-msvc": "0.150.0", + "@oxc-parser/binding-win32-x64-msvc": "0.150.0" + } + }, + "node_modules/prettier": { + "version": "3.8.1", + "resolved": "https://registry.npmjs.org/prettier/-/prettier-3.8.1.tgz", + "integrity": "sha512-UOnG6LftzbdaHZcKoPFtOcCKztrQ57WkHDeRD9t/PTQtmT0NHSeWWepj6pS0z/N7+08BHFDQVUrfmfMRcZwbMg==", + "dev": true, + "license": "MIT", + "bin": { + "prettier": "bin/prettier.cjs" + }, + "engines": { + "node": ">=14" + }, + "funding": { + "url": "https://github.com/prettier/prettier?sponsor=1" + } + }, + "node_modules/typescript": { + "version": "5.9.3", + "resolved": "https://registry.npmjs.org/typescript/-/typescript-5.9.3.tgz", + "integrity": "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw==", + "dev": true, + "license": "Apache-2.0", + "bin": { + "tsc": "bin/tsc", + "tsserver": "bin/tsserver" + }, + "engines": { + "node": ">=14.17" + } + }, + "node_modules/undici-types": { + "version": "7.18.2", + "resolved": "https://registry.npmjs.org/undici-types/-/undici-types-7.18.2.tgz", + "integrity": "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w==", + "dev": true, + "license": "MIT" + } + } +} diff --git a/engine/package.json b/engine/package.json new file mode 100644 index 000000000..956b89a26 --- /dev/null +++ b/engine/package.json @@ -0,0 +1,34 @@ +{ + "name": "@logic-engine/compiler", + "version": "0.1.0", + "private": true, + "description": "Single-source logic engine: author a library once in a restricted, semantically typed subset of TypeScript and generate native, idiomatic implementations for TypeScript, Python and Go.", + "license": "MIT", + "type": "module", + "engines": { + "node": ">=22.12.0" + }, + "bin": { + "logic-engine": "./src/cli.ts" + }, + "exports": { + ".": "./src/api.ts", + "./prelude": "./prelude/index.d.ts" + }, + "scripts": { + "build": "node ./src/cli.ts build", + "test": "node --test --test-reporter=dot tests/*.spec.ts", + "verify": "node ./scripts/verify.ts", + "docs": "node ./scripts/intrinsics-doc.ts", + "metrics": "node ./scripts/metrics.ts" + }, + "dependencies": { + "oxc-parser": "0.150.0" + }, + "devDependencies": { + "@types/node": "24.13.6", + "esbuild": "0.27.4", + "prettier": "3.8.1", + "typescript": "5.9.3" + } +} diff --git a/engine/prelude/index.d.ts b/engine/prelude/index.d.ts new file mode 100644 index 000000000..2d23af826 --- /dev/null +++ b/engine/prelude/index.d.ts @@ -0,0 +1,203 @@ +/** + * The authoring prelude: declarations only, never executed. + * + * `tsc` sees `Int` and `Digits` as `number` and `string`, so the editor, go-to-definition and + * inline errors all work on ordinary TypeScript. The engine's own checker sees them as distinct + * semantic types with ranges and refinements. Source files are never run: only generated code is. + */ + +/** A mathematical integer. Its proven range is inferred; annotate with `IntRange` to pin it. */ +declare type Int = number; + +/** An integer proven to lie in `[Lo, Hi]`. The bounds must be literals. */ +declare type IntRange = number; + +/** IEEE-754 binary64. */ +declare type Float = number; + +/** An exact decimal with a fixed scale. */ +declare type Decimal = { readonly __decimal: Scale }; + +/** A string proven to hold only scalars below U+0080. */ +declare type Ascii = string; + +/** A string proven to hold only ASCII digits. */ +declare type Digits = string; + +/** An ASCII string of exactly `N` scalars. */ +declare type AsciiOf = string; + +/** A digit string of exactly `N` scalars. */ +declare type DigitsOf = string; + +/** A list whose length is proven to lie in `[Min, Max]`. */ +declare type List = readonly T[]; + +/** A date on the proleptic Gregorian calendar, years 1 to 9999, with no zone. */ +declare type CivilDate = { readonly __civilDate: unique symbol }; + +/** Milliseconds since the Unix epoch. */ +declare type Instant = { readonly __instant: unique symbol }; + +/** An exact count of milliseconds. */ +declare type Duration = { readonly __duration: unique symbol }; + +/** The rounding mode every lossy decimal operation names explicitly. */ +declare type RoundingMode = + | "half-even" + | "half-up" + | "half-down" + | "down" + | "up" + | "ceil" + | "floor"; + +/** The root of every domain error. Subclasses must have empty bodies. */ +declare class DomainError extends Error {} + +/** Raised by `http.request` on a transport error or a timeout. */ +declare class HttpError extends DomainError {} + +declare type HttpHeader = { readonly name: Ascii; readonly value: string }; + +declare type HttpRequest = { + readonly method: Ascii; + readonly url: string; + readonly headers: readonly HttpHeader[]; + readonly body: string; + readonly timeoutMillis: Int; +}; + +declare type HttpResponse = { + readonly status: Int; + readonly headers: readonly HttpHeader[]; + readonly body: string; +}; + +declare namespace str { + function len(value: string): Int; + function codePoints(value: string): readonly Int[]; + function fromCodePoints(points: readonly Int[]): string; + function concat(left: string, right: string): string; + function codeAt(value: Ascii, index: Int): Int; + function charAt(value: Ascii, index: Int): Ascii; + function charAtOpt(value: Ascii, index: Int): Ascii | undefined; + function codeAtOpt(value: Ascii, index: Int): Int | undefined; + function slice(value: Ascii, from: Int, to: Int): Ascii; + function indexOf(value: string, needle: string): Int; + function contains(value: string, needle: string): boolean; + function startsWith(value: string, prefix: string): boolean; + function endsWith(value: string, suffix: string): boolean; + function repeat(value: string, count: Int): string; + function padStart(value: string, length: Int, pad: string): string; + function trim(value: string): string; + function asciiUpper(value: Ascii): Ascii; + function asciiLower(value: Ascii): Ascii; + function compare(left: string, right: string): Int; + function asAscii(value: string): Ascii | undefined; + function asDigits(value: string): Digits | undefined; + function split(value: string, separator: Ascii): readonly string[]; + function join(values: readonly string[], separator: string): string; + function fromInt(value: Int): Ascii; + function parseInt(value: string): Int | undefined; +} + +declare namespace seq { + function len(list: readonly T[]): Int; + function get(list: readonly T[], index: Int): T; + function at(list: readonly T[], index: Int): T | undefined; + function map(list: readonly T[], fn: (item: T) => R): readonly R[]; + function filter(list: readonly T[], fn: (item: T) => boolean): readonly T[]; + function fold(list: readonly T[], initial: A, fn: (accumulator: A, item: T) => A): A; + function sum(list: readonly Int[]): Int; + function any(list: readonly T[], fn: (item: T) => boolean): boolean; + function all(list: readonly T[], fn: (item: T) => boolean): boolean; + function find(list: readonly T[], fn: (item: T) => boolean): T | undefined; + function indexOf(list: readonly T[], needle: T): Int; + function contains(list: readonly T[], needle: T): boolean; + function concat(left: readonly T[], right: readonly T[]): readonly T[]; + function slice(list: readonly T[], from: Int, to: Int): readonly T[]; + function reverse(list: readonly T[]): readonly T[]; + function sortStable(list: readonly T[], compare: (left: T, right: T) => Int): readonly T[]; + function sortStableBy(list: readonly T[], key: (item: T) => K): readonly T[]; +} + +declare namespace re { + /** Whole-string match against a comptime pattern. */ + function test(pattern: RegExp, value: string): boolean; + /** Keeps only the scalars matching a comptime character class. */ + function retain(pattern: RegExp, value: string): string; +} + +declare namespace int { + function abs(value: Int): Int; + function min(left: Int, right: Int): Int; + function max(left: Int, right: Int): Int; +} + +declare namespace float { + function fromInt(value: Int): Float; +} + +declare namespace dec { + function fromScaled(unscaled: Int, scale: S): Decimal; + function fromInt(value: Int, scale: S): Decimal; + function fromFloat(value: Float, scale: S, mode: RoundingMode): Decimal; + function add(left: Decimal, right: Decimal): Decimal; + function sub(left: Decimal, right: Decimal): Decimal; + function mul(left: Decimal, right: Decimal): Decimal; + function divRound( + left: Decimal, + right: Decimal, + scale: S, + mode: RoundingMode, + ): Decimal; + function rescale(value: Decimal, scale: S, mode: RoundingMode): Decimal; + function compare(left: Decimal, right: Decimal): Int; + function isNegative(value: Decimal): boolean; + function abs(value: Decimal): Decimal; + function unscaled(value: Decimal): Int; +} + +declare namespace date { + function fromYmd(year: Int, month: Int, day: Int): CivilDate | undefined; + function fromEpochDays(days: Int): CivilDate | undefined; + function clampEpochDays(days: Int): CivilDate; + function toEpochDays(value: CivilDate): Int; + function year(value: CivilDate): Int; + function month(value: CivilDate): Int; + function day(value: CivilDate): Int; + function addDays(value: CivilDate, days: Int): CivilDate | undefined; + function diffDays(left: CivilDate, right: CivilDate): Int; + function dayOfWeek(value: CivilDate): Int; + function compare(left: CivilDate, right: CivilDate): Int; + function isLeapYear(year: Int): boolean; +} + +declare namespace opt { + function isNone(value: T | undefined): boolean; + function unwrap(value: T | undefined): T; + function some(value: T): T | undefined; + function orElse(value: T | undefined, fallback: T): T; +} + +declare namespace http { + function request(request: HttpRequest): HttpResponse; +} + +declare namespace clock { + function now(): Instant; + function sleep(duration: Duration): void; + function millis(value: Int): Duration; + function elapsed(from: Instant, to: Instant): Duration; + function durationMillis(duration: Duration): Int; +} + +declare namespace random { + function nextU32(): Int; +} + +declare namespace task { + /** Runs idempotent tasks concurrently and takes the first success. */ + function race(tasks: readonly (() => T)[]): T | undefined; +} diff --git a/engine/scripts/fuzz.ts b/engine/scripts/fuzz.ts new file mode 100644 index 000000000..a10e8f30e --- /dev/null +++ b/engine/scripts/fuzz.ts @@ -0,0 +1,115 @@ +#!/usr/bin/env node +/** + * The random program generator's command line. + * + * node scripts/fuzz.ts fast [--seed N] [--count N] [--cases N] + * node scripts/fuzz.ts full [--seed N] [--count N] [--cases N] [--targets ts,python,go,rust] + * + * `fast` compiles each generated program and checks the reference interpreter's actual answers + * against the range, length and character class the checker proved for them — no code generation, + * so it is cheap enough to run in the thousands on every `verify` (see `../docs/fuzzing.md`). + * + * `full` additionally generates all four targets, in both idiom modes, and compares every answer + * against the interpreter's, reusing `src/conformance/differential.ts` exactly as + * `core/conformance/run.ts` does by hand. It is slow (four toolchains, twice each) and is meant to + * be run deliberately, not on every commit. + * + * Either mode exits non-zero when it finds a divergence, and prints the seed plus the shrunken + * source so the failure can be replayed with `--seed --count 1`. + */ + +import { runFast, runFull, shrinkViolation } from "../src/fuzz/harness.ts"; + +function flag(name: string, fallback: string): string { + const index = process.argv.indexOf(`--${name}`); + return index === -1 ? fallback : (process.argv[index + 1] ?? fallback); +} + +/** JSON cannot carry a bigint on its own; this is only for printing a reproducing input. */ +function jsonish(value: unknown): unknown { + if (typeof value === "bigint") return value.toString(); + if (Array.isArray(value)) return value.map(jsonish); + if (typeof value === "object" && value !== null) { + return Object.fromEntries(Object.entries(value as Record).map(([k, v]) => [k, jsonish(v)])); + } + return value; +} + +function main(): void { + const mode = process.argv[2]; + const seed = Number(flag("seed", `${Date.now() >>> 0}`)); + const count = Number(flag("count", "500")); + const cases = Number(flag("cases", mode === "full" ? "4" : "8")); + + if (mode === "fast") { + process.stdout.write(`fast mode: seed=${seed} count=${count} cases/program=${cases}\n`); + let lastReported = 0; + const report = runFast(seed, count, cases, (attempted) => { + if (attempted - lastReported >= 500) { + process.stdout.write(` ${attempted}/${count}…\n`); + lastReported = attempted; + } + }); + process.stdout.write( + `attempted ${report.attempted}, compiled ${report.compiled} ` + + `(${((report.compiled / report.attempted) * 100).toFixed(1)}%), ` + + `ran ${report.casesRun} cases in ${(report.elapsedMs / 1000).toFixed(1)}s\n`, + ); + if (report.violations.length === 0) { + process.stdout.write("ok: every produced value stayed inside its proven bounds\n"); + return; + } + process.stderr.write(`\n${report.violations.length} bounds violation(s) found — shrinking…\n\n`); + for (const violation of report.violations) { + const shrunk = shrinkViolation(violation, cases); + process.stderr.write(`seed ${seed} (program #${violation.index}, replay: --seed ${seed} --count 1):\n`); + process.stderr.write(`${shrunk.source}\n`); + process.stderr.write(`input: ${JSON.stringify(shrunk.input.map(jsonish))}\n`); + process.stderr.write( + `the interpreter produced ${JSON.stringify(jsonish(violation.produced))}, which is not a ` + + `${violation.retType} the checker's proven bound admits\n\n`, + ); + } + process.exitCode = 1; + return; + } + + if (mode === "full") { + const targetsFlag = flag("targets", "typescript,python,go,rust"); + const targets = targetsFlag.split(",").map((t) => t.trim()) as ("typescript" | "python" | "go" | "rust")[]; + process.stdout.write( + `full mode: seed=${seed} count=${count} cases/program=${cases} targets=${targets.join(",")}\n`, + ); + let lastReported = 0; + const report = runFull(seed, count, cases, targets, (attempted) => { + if (attempted - lastReported >= 100) { + process.stdout.write(` prefilter ${attempted}/${count}…\n`); + lastReported = attempted; + } + }); + process.stdout.write( + `attempted ${report.attempted}, compiled ${report.compiled} ` + + `(${((report.compiled / report.attempted) * 100).toFixed(1)}%), ` + + `${(report.elapsedMs / 1000).toFixed(1)}s\n`, + ); + if (report.divergences.length === 0) { + process.stdout.write("ok: interpreter and every target agree, in both idiom modes\n"); + return; + } + process.stderr.write(`\n${report.divergences.length} divergence(s) found\n\n`); + for (const divergence of report.divergences) { + process.stderr.write(`seed ${seed}, target ${divergence.target}, case fn=${divergence.case.fn}:\n`); + process.stderr.write(`${divergence.source}\n`); + process.stderr.write(`args: ${JSON.stringify(divergence.case.args.map(jsonish))}\n`); + process.stderr.write(`expected: ${JSON.stringify(divergence.expected)}\n`); + process.stderr.write(`actual: ${JSON.stringify(divergence.actual)}\n\n`); + } + process.exitCode = 1; + return; + } + + process.stderr.write("usage: node scripts/fuzz.ts fast|full [--seed N] [--count N] [--cases N]\n"); + process.exitCode = 1; +} + +main(); diff --git a/engine/scripts/intrinsics-doc.ts b/engine/scripts/intrinsics-doc.ts new file mode 100644 index 000000000..81656e3ef --- /dev/null +++ b/engine/scripts/intrinsics-doc.ts @@ -0,0 +1,46 @@ +#!/usr/bin/env node +/** + * Writes `docs/intrinsics.md` from the registry, so the specification of the operation set cannot + * drift from the operations the compiler actually has. + */ + +import { writeFileSync } from "node:fs"; +import { join } from "node:path"; +import { allIntrinsics } from "../src/intrinsics/index.ts"; +import { effectsToString } from "../src/effects.ts"; + +const groups = new Map(); + +for (const intrinsic of allIntrinsics()) { + const module = intrinsic.name.split(".")[0]!; + const rows = groups.get(module) ?? []; + rows.push({ + name: intrinsic.name, + doc: intrinsic.doc, + effects: effectsToString(intrinsic.effects), + comptime: intrinsic.comptime, + }); + groups.set(module, rows); +} + +const lines = [ + "# Intrinsics", + "", + "Generated by `npm run docs`. One row per operation the Core can express; everything else is", + "source library code, which is the default answer (see the admission rule in", + "[semantics.md](semantics.md)).", + "", + `There are ${allIntrinsics().length} intrinsics in ${groups.size} modules.`, + "", +]; + +for (const [module, rows] of [...groups.entries()].sort(([left], [right]) => left.localeCompare(right))) { + lines.push(`## \`${module}\``, "", "| operation | effects | comptime | meaning |", "| --- | --- | --- | --- |"); + for (const row of rows.sort((left, right) => left.name.localeCompare(right.name))) { + lines.push(`| \`${row.name}\` | ${row.effects} | ${row.comptime ? "yes" : "no"} | ${row.doc} |`); + } + lines.push(""); +} + +writeFileSync(join(import.meta.dirname, "..", "docs", "intrinsics.md"), `${lines.join("\n")}\n`); +process.stdout.write(`docs/intrinsics.md: ${allIntrinsics().length} intrinsics\n`); diff --git a/engine/scripts/metrics.ts b/engine/scripts/metrics.ts new file mode 100644 index 000000000..96b6e17c4 --- /dev/null +++ b/engine/scripts/metrics.ts @@ -0,0 +1,131 @@ +#!/usr/bin/env node +/** + * The metrics the architecture is judged by, computed rather than asserted. + * + * Usage: `node scripts/metrics.ts [--json]`. Everything here is mechanical: the numbers + * come from the compiler and from the generated files, so a claim in `progress.md` can always be + * re-derived. + */ + +import { readFileSync, readdirSync, statSync } from "node:fs"; +import { join, relative, resolve } from "node:path"; +import { compileProject } from "../src/api.ts"; +import { generate } from "../src/backend/generate.ts"; +import { GO_BACKEND } from "../src/targets/go/index.ts"; +import { PYTHON_BACKEND } from "../src/targets/python/index.ts"; +import { TYPESCRIPT_BACKEND } from "../src/targets/typescript/index.ts"; +import { RUST_BACKEND } from "../src/targets/rust/index.ts"; +import { typeToString } from "../src/types.ts"; + +const ENGINE = resolve(import.meta.dirname, ".."); +const project = resolve(process.argv[2] ?? "."); + +function linesOf(path: string): number { + return readFileSync(path, "utf8").split("\n").filter((line) => line.trim() !== "").length; +} + +function linesUnder(root: string, filter: (path: string) => boolean = () => true): number { + let total = 0; + const walk = (directory: string): void => { + for (const entry of readdirSync(directory)) { + const full = join(directory, entry); + if (statSync(full).isDirectory()) walk(full); + else if (filter(full)) total += linesOf(full); + } + }; + walk(root); + return total; +} + +const compilation = compileProject(join(project, "source")); +const backends = [TYPESCRIPT_BACKEND, PYTHON_BACKEND, GO_BACKEND, RUST_BACKEND]; + +const lowering: Record> = {}; +const generatedLines: Record = {}; +const perUtility: Record> = {}; + +for (const backend of backends) { + const result = generate(compilation.program, backend); + const counts = { native: 0, library: 0, portable: 0 }; + const seen = new Set(); + for (const selection of result.selections) { + const key = `${selection.op}(${selection.args})`; + if (seen.has(key)) continue; + seen.add(key); + counts[selection.impl as keyof typeof counts] += 1; + } + lowering[backend.spec.name] = counts; + generatedLines[backend.spec.name] = result.files + .filter((file) => !file.path.startsWith("_driver") && !file.path.includes("cmd/")) + .reduce((total, file) => total + file.text.split("\n").filter((line) => line.trim() !== "").length, 0); + + for (const file of result.files) { + const utility = file.path.replace(/\.[a-z]+$/, ""); + if (!compilation.program.entryPoints.some((name) => name.split("::")[0] === utility.replaceAll("_", "-"))) continue; + perUtility[utility] = perUtility[utility] ?? {}; + perUtility[utility]![backend.spec.name] = file.text.split("\n").filter((line) => line.trim() !== "").length; + } +} + +const wideIntegers = new Set(); +for (const fn of compilation.program.functions.values()) { + for (const param of fn.params) { + const rendered = typeToString(param.type); + if (param.type.kind === "Int" && (param.type.lo < -(2n ** 53n - 1n) || param.type.hi > 2n ** 53n - 1n)) { + wideIntegers.add(`${fn.name}(${param.name}: ${rendered})`); + } + } +} + +const metrics = { + utilities: compilation.program.entryPoints.length, + coreFunctions: compilation.program.functions.size, + sourceLines: linesUnder(join(project, "source")) + linesUnder(join(ENGINE, "stdlib")), + compilerLines: { + frontend: linesUnder(join(ENGINE, "src", "frontend")) + linesUnder(join(ENGINE, "src", "hir")), + core: linesUnder(join(ENGINE, "src", "core")), + analysisAndPasses: + linesUnder(join(ENGINE, "src", "analysis")) + + linesUnder(join(ENGINE, "src", "optimize")) + + linesUnder(join(ENGINE, "src", "link")) + + linesUnder(join(ENGINE, "src", "comptime")), + interpreter: linesUnder(join(ENGINE, "src", "interp")), + intrinsics: linesUnder(join(ENGINE, "src", "intrinsics")), + backendFramework: linesUnder(join(ENGINE, "src", "backend")), + targets: Object.fromEntries( + backends.map((backend) => [ + backend.spec.name, + linesUnder(join(ENGINE, "src", "targets", backend.spec.name)), + ]), + ), + }, + lowering, + generatedLines, + perUtility, + wideIntegers: [...wideIntegers], + widenedLoops: compilation.metrics.widenedLoops, + clampedRanges: compilation.metrics.clampedRanges.length, +}; + +if (process.argv.includes("--json")) { + process.stdout.write(`${JSON.stringify(metrics, null, "\t")}\n`); +} else { + const targetLines = Object.values(metrics.compilerLines.targets).reduce((a, b) => a + b, 0); + const frontendCoreAnalysis = + metrics.compilerLines.frontend + metrics.compilerLines.core + metrics.compilerLines.analysisAndPasses; + process.stdout.write( + [ + `utilities: ${metrics.utilities}`, + `core functions: ${metrics.coreFunctions}`, + `source lines: ${metrics.sourceLines}`, + `frontend+core+analysis: ${frontendCoreAnalysis}`, + `backends (4 targets): ${targetLines}`, + `generated lines: ${Object.entries(metrics.generatedLines).map(([name, count]) => `${name} ${count}`).join(", ")}`, + `lowering mix: ${Object.entries(lowering).map(([name, counts]) => `${name} ${counts["native"]}n/${counts["library"]}l/${counts["portable"]}p`).join(", ")}`, + `wide integers: ${metrics.wideIntegers.length}`, + `widened loops: ${metrics.widenedLoops}`, + `clamped ranges: ${metrics.clampedRanges}`, + `relative path: ${relative(ENGINE, project)}`, + ].join("\n") + "\n", + ); +} diff --git a/engine/scripts/size.ts b/engine/scripts/size.ts new file mode 100644 index 000000000..7c48d0e20 --- /dev/null +++ b/engine/scripts/size.ts @@ -0,0 +1,187 @@ +#!/usr/bin/env node +/** + * What a consumer of the generated TypeScript actually pays, per utility. + * + * The npm package this engine generates for is tree-shakeable, and ADR 0012 records that as a + * requirement rather than a preference: a consumer who imports `isValidCpf` must not carry + * `getHolidays`. So the TypeScript target's cost is not only nanoseconds — it is bytes over the + * wire — and a cost that is never measured is a cost that is asserted. + * + * This measures it the way a consumer's bundler would: one single-import entry point per exported + * utility, bundled and minified by esbuild against the generated tree, then gzipped. Raw source + * bytes are the wrong number (comments, types and formatting all vanish before a browser sees + * them), and the whole tree is the wrong number too (nobody imports all of it). + * + * Usage: + * node engine/scripts/size.ts + * Print the table. Nothing is written: the committed `SIZE.json` is a baseline, and a baseline + * that moves whenever it is looked at is not one. + * + * node engine/scripts/size.ts --check + * Compare against the committed `SIZE.json` and exit 1 when any export grew at all. This is + * what `verify` runs: the trade is checked, not remembered. + * + * node engine/scripts/size.ts --write + * Accept the current measurement as the new baseline. + */ + +import { mkdtempSync, readFileSync, rmSync, writeFileSync, existsSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join, resolve } from "node:path"; +import { brotliCompressSync, constants, gzipSync } from "node:zlib"; + +import { build } from "esbuild"; + +/** + * How many gzipped bytes a pre-existing export may grow before it counts as a regression. + * + * Zero, deliberately. Every other target's optimizations are paid for in the currency that target + * is judged in, and for this one that currency is bytes a browser downloads: an optimization here + * is only an optimization if the bundle does not grow for it. The number is not a tolerance to be + * widened when something gets close — a change that needs more room is a change whose trade has to + * be argued, measured and written down, and then accepted with `--write`. + * + * Measured on gzipped output, so it is also not noisy: the same input produces the same bytes. + */ +const GROWTH_BYTES = 0; + +/** + * Three numbers, because they answer different questions and this project has been wrong about + * which one matters before. + * + * `minified` is what the browser parses and what the JavaScript engine holds — it is not the + * transfer size, but it is the only one of the three that tracks parse and compile cost. + * `gzip` and `brotli` are both transfer sizes, and they disagree: gzip's window makes locally + * repeated text almost free, so a shorter but less repetitive encoding can be smaller raw and + * larger gzipped. Brotli's larger window and its static dictionary of common web text weigh the + * same source differently, and brotli is what most CDNs actually serve. Both are gated, so a + * change has to be no worse under either. + */ +type Measurement = { minified: number; gzip: number; brotli: number }; +type Snapshot = { exports: Record; total: Measurement }; + +type ApiFunction = { name: string; module: string; effects: readonly string[] }; + +function exportsOf(outDir: string): ApiFunction[] { + const api = JSON.parse(readFileSync(join(outDir, "API.json"), "utf8")) as { functions: ApiFunction[] }; + // One row per public entry point, which is what a consumer writes. The `…With` seam (ADR 0011) + // is reachable from the same module and is measured by whichever of its callers pulls it in. + return [...api.functions].sort((a, b) => a.name.localeCompare(b.name)); +} + +async function measure(outDir: string, entries: readonly { name: string; module: string }[]): Promise { + const scratch = mkdtempSync(join(tmpdir(), "engine-size-")); + try { + const imports = entries + .map((entry, index) => `import { ${entry.name} as e${index} } from ${JSON.stringify(join(outDir, entry.module))};`) + .join("\n"); + const uses = entries.map((_, index) => `e${index}`).join(", "); + // `globalThis.__keep` is an escape the optimizer cannot see through, so nothing measured here + // is dropped for being unobserved — only for being genuinely unreachable from the import. + const entryPath = join(scratch, "entry.ts"); + writeFileSync(entryPath, `${imports}\nglobalThis.__keep = [${uses}];\n`); + const result = await build({ + entryPoints: [entryPath], + bundle: true, + minify: true, + format: "esm", + platform: "browser", + target: "es2020", + treeShaking: true, + write: false, + legalComments: "none", + }); + const code = result.outputFiles[0]!.contents; + return { + minified: code.byteLength, + gzip: gzipSync(code, { level: 9 }).byteLength, + // Quality 11 and a size hint, because the defaults are tuned for streaming and would + // understate what a CDN serving a static asset produces. + brotli: brotliCompressSync(code, { + params: { + [constants.BROTLI_PARAM_QUALITY]: 11, + [constants.BROTLI_PARAM_SIZE_HINT]: code.byteLength, + }, + }).byteLength, + }; + } finally { + rmSync(scratch, { recursive: true, force: true }); + } +} + +function formatBytes(value: number): string { + return value.toLocaleString("en-US"); +} + +async function main(): Promise { + const project = resolve(process.argv[2] ?? "."); + const mode = process.argv.includes("--check") ? "check" : process.argv.includes("--write") ? "write" : "print"; + const outDir = join(project, "out", "typescript"); + if (!existsSync(join(outDir, "API.json"))) { + process.stdout.write(`skipped: no generated TypeScript at ${outDir}\n`); + return; + } + + const api = exportsOf(outDir); + const rows = await Promise.all( + api.map(async (fn) => ({ name: fn.name, measurement: await measure(outDir, [fn]) })), + ); + const total = await measure(outDir, api); + + const snapshot: Snapshot = { + exports: Object.fromEntries(rows.map((row) => [row.name, row.measurement])), + total, + }; + + const namePad = Math.max(8, ...rows.map((row) => row.name.length)); + const line = (name: string, measurement: Measurement): string => + `${name.padEnd(namePad)} ${formatBytes(measurement.minified).padStart(10)} ${formatBytes(measurement.gzip).padStart(8)} ${formatBytes(measurement.brotli).padStart(8)}\n`; + process.stdout.write( + `${"export".padEnd(namePad)} ${"minified".padStart(10)} ${"gzip".padStart(8)} ${"brotli".padStart(8)}\n`, + ); + for (const row of rows) process.stdout.write(line(row.name, row.measurement)); + process.stdout.write(line("(all)", total)); + + const snapshotPath = join(outDir, "SIZE.json"); + if (mode === "check") { + if (!existsSync(snapshotPath)) { + process.stdout.write("no SIZE.json to compare against\n"); + process.exitCode = 1; + return; + } + const base = JSON.parse(readFileSync(snapshotPath, "utf8")) as Snapshot; + const regressions: string[] = []; + // Both transfer encodings, because a consumer gets whichever their CDN negotiates and the + // two do not agree on which source is smaller. `minified` is reported but not gated: it is + // a parse cost, not a transfer cost, and a change that trades it against a transfer size is + // a trade to argue rather than a threshold to trip. + const compared = ["gzip", "brotli"] as const; + const check = (label: string, previous: Measurement, current: Measurement): void => { + for (const metric of compared) { + const grew = current[metric] - previous[metric]; + if (grew > GROWTH_BYTES) { + regressions.push( + `${label}: ${formatBytes(previous[metric])} -> ${formatBytes(current[metric])} ${metric} (+${formatBytes(grew)})`, + ); + } + } + }; + for (const [name, measurement] of Object.entries(snapshot.exports)) { + const previous = base.exports[name]; + if (previous === undefined) continue; + check(name, previous, measurement); + } + check("(all)", base.total, snapshot.total); + if (regressions.length > 0) { + process.stdout.write(`\n${regressions.length} size regression(s):\n${regressions.map((line) => ` ${line}`).join("\n")}\n`); + process.exitCode = 1; + } + return; + } + + if (mode !== "write") return; + writeFileSync(snapshotPath, `${JSON.stringify(snapshot, undefined, "\t")}\n`); + process.stdout.write(`\nwrote ${snapshotPath}\n`); +} + +await main(); diff --git a/engine/scripts/verify.ts b/engine/scripts/verify.ts new file mode 100644 index 000000000..2acbf9599 --- /dev/null +++ b/engine/scripts/verify.ts @@ -0,0 +1,228 @@ +#!/usr/bin/env node +/** + * `verify` for a project: the one command the milestones are measured by. + * + * It compiles, generates every target in both idiom modes, regenerates and diffs for determinism, + * runs each target's linters, and then runs the project's own conformance runner. A milestone is + * not complete while this is red. + */ + +import { spawnSync } from "node:child_process"; +import { cpSync, existsSync, readFileSync, readdirSync, rmSync, statSync } from "node:fs"; +import { join, relative, resolve } from "node:path"; +import { formattersOf } from "../src/backend/format.ts"; + +const ENGINE = resolve(import.meta.dirname, ".."); +const project = resolve(process.argv[2] ?? "."); + +type Step = { name: string; run: () => { ok: boolean; output?: string } }; + +function shell(command: string, args: readonly string[], cwd: string): { ok: boolean; output?: string } { + const result = spawnSync(command, [...args], { cwd, encoding: "utf8" }); + if (result.error !== undefined && (result.error as NodeJS.ErrnoException).code === "ENOENT") { + return { ok: true, output: `skipped: ${command} is not installed` }; + } + return { ok: result.status === 0, output: `${result.stdout}${result.stderr}`.trim() }; +} + +function filesOf(root: string): Map { + const files = new Map(); + if (!existsSync(root)) return files; + const walk = (directory: string): void => { + for (const entry of readdirSync(directory)) { + const full = join(directory, entry); + if (statSync(full).isDirectory()) walk(full); + else files.set(relative(root, full), readFileSync(full, "utf8")); + } + }; + walk(root); + return files; +} + +/** Every generated TypeScript file except the differential driver, which imports node globals. */ +function generatedTypeScript(root: string): string[] { + const files: string[] = []; + const walk = (directory: string): void => { + for (const entry of readdirSync(directory)) { + const full = join(directory, entry); + if (statSync(full).isDirectory()) walk(full); + else if (entry.endsWith(".ts") && entry !== "_driver.ts") files.push(relative(root, full)); + } + }; + walk(root); + return files; +} + +const steps: Step[] = [ + { + // The generated output is committed, so a missing formatter is not a missing nicety: it + // produces different bytes for the same program, and every later step would pass while the + // checkout drifts from what is in the repository. + name: "formatters", + run: () => { + const missing = ["typescript", "python", "go", "rust"].flatMap((target) => + formattersOf(target) + .filter((formatter) => !formatter.installed) + .map((formatter) => `${target}: ${formatter.name}`), + ); + return missing.length === 0 + ? { ok: true } + : { ok: false, output: `not installed, so the output would differ from the committed one:\n${missing.join("\n")}` }; + }, + }, + { + name: "engine tests", + run: () => + shell( + process.execPath, + [ + "--test", + "--test-reporter=dot", + ...readdirSync(join(ENGINE, "tests")) + .filter((entry) => entry.endsWith(".spec.ts")) + .map((entry) => join("tests", entry)), + ], + ENGINE, + ), + }, + { + name: "engine typecheck", + run: () => shell(join(ENGINE, "node_modules", ".bin", "tsc"), ["--noEmit", "-p", "tsconfig.json"], ENGINE), + }, + { + // A small, fixed seed budget on every verification: no code generation, so it is cheap + // enough to run here rather than only on demand. See docs/fuzzing.md for the full story, + // and `node scripts/fuzz.ts full` for the slower, deliberately-run four-target comparison. + name: "fuzz (fast, checker vs. interpreter)", + run: () => shell(process.execPath, [join(ENGINE, "scripts", "fuzz.ts"), "fast", "--seed", "20260921", "--count", "1000"], ENGINE), + }, + { + name: "check", + run: () => shell(process.execPath, [join(ENGINE, "src", "cli.ts"), "check", "--project", project], project), + }, + { + name: "generate (idiomatic)", + run: () => shell(process.execPath, [join(ENGINE, "src", "cli.ts"), "build", "--project", project], project), + }, + { + name: "generate (--no-idioms)", + run: () => + shell(process.execPath, [join(ENGINE, "src", "cli.ts"), "build", "--project", project, "--no-idioms"], project), + }, + { + name: "determinism", + run: () => { + const out = join(project, "out"); + const copy = join(project, ".out-previous"); + rmSync(copy, { recursive: true, force: true }); + cpSync(out, copy, { recursive: true }); + const regenerated = shell(process.execPath, [join(ENGINE, "src", "cli.ts"), "build", "--project", project], project); + if (!regenerated.ok) return regenerated; + const before = filesOf(copy); + const after = filesOf(out); + rmSync(copy, { recursive: true, force: true }); + for (const [path, text] of after) { + // Fixtures are written by the conformance runner, and formatter caches by the + // formatters; neither is compiler output. + if (path.endsWith("fixtures.json") || path.includes("cache")) continue; + if (before.get(path) !== text) return { ok: false, output: `${path} differs between two runs` }; + } + return { ok: true }; + }, + }, + { + name: "typescript typecheck", + run: () => + shell( + join(ENGINE, "node_modules", ".bin", "tsc"), + [ + "--noEmit", + "--strict", + "--target", + "es2022", + // The generated capability defaults use the platform's own fetch, timers and crypto, + // which live in the DOM library and are global in Node 20 and later. + "--lib", + "es2022,dom", + "--module", + "nodenext", + "--moduleResolution", + "nodenext", + "--allowImportingTsExtensions", + "--skipLibCheck", + ...generatedTypeScript(join(project, "out", "typescript")), + ], + join(project, "out", "typescript"), + ), + }, + { + // The generated TypeScript is shipped to a browser by a tree-shakeable package, so its size + // is a result this pipeline has to check rather than a trade to remember. `size.ts` measures + // what a consumer's bundler would produce for a single-import entry point and compares it + // with the committed `SIZE.json`; a per-export regression past the budget fails here. + name: "typescript size", + run: () => shell(process.execPath, [join(ENGINE, "scripts", "size.ts"), project, "--check"], project), + }, + { + name: "python compile", + run: () => shell("python3", ["-m", "compileall", "-q", "."], join(project, "out", "python")), + }, + { + name: "go vet", + run: () => shell("go", ["vet", "./..."], join(project, "out", "go")), + }, + { + name: "rust build", + run: () => shell("cargo", ["build", "--offline", "--release"], join(project, "out", "rust")), + }, + { + name: "rust clippy", + run: () => shell("cargo", ["clippy", "--offline", "--", "-D", "warnings"], join(project, "out", "rust")), + }, + { + name: "rust fmt check", + run: () => shell("cargo", ["fmt", "--check"], join(project, "out", "rust")), + }, + { + name: "conformance", + run: () => { + const runner = join(project, "conformance", "run.ts"); + if (!existsSync(runner)) return { ok: true, output: "skipped: no conformance runner" }; + return shell( + process.execPath, + ["--import", join(project, "conformance", "sloppy-imports.mjs"), runner], + project, + ); + }, + }, + { + // Every step above proves the source *compiles* — this proves it *runs*: the migrated + // utilities under `source/`, imported and called in plain Node, no engine involved. See + // `conformance/run-source.ts` for exactly which utilities that covers and why the rest + // cannot run yet. + name: "conformance (source, no engine)", + run: () => { + const runner = join(project, "conformance", "run-source.ts"); + if (!existsSync(runner)) return { ok: true, output: "skipped: no source conformance runner" }; + return shell( + process.execPath, + ["--import", join(project, "conformance", "sloppy-imports.mjs"), runner], + project, + ); + }, + }, +]; + +let failed = false; +for (const step of steps) { + const result = step.run(); + const status = result.ok ? "ok" : "FAILED"; + process.stdout.write(`${status.padEnd(7)} ${step.name}\n`); + if (!result.ok || (result.output ?? "").startsWith("skipped")) { + const output = (result.output ?? "").trim(); + if (output !== "") process.stdout.write(`${output.split("\n").map((line) => ` ${line}`).join("\n")}\n`); + } + if (!result.ok) failed = true; +} + +if (failed) process.exitCode = 1; diff --git a/engine/src/analysis/borrows.ts b/engine/src/analysis/borrows.ts new file mode 100644 index 000000000..21002a42c --- /dev/null +++ b/engine/src/analysis/borrows.ts @@ -0,0 +1,276 @@ +/** + * Borrow inference: a whole-program pre-pass over the Core that decides, for every String- or + * List-typed function parameter, whether it may be printed as a borrow (`&str`/`&[T]`) instead of + * an owned value (`String`/`Vec`). + * + * This runs once, over the full `CProgram`, before any target-specific lowering or printing + * starts — `lowerProgram` receives every function in the program at once, unlike a candidate's + * `emit` (which runs before any function scope exists) or a backend's `printModule` (which sees + * one module at a time). See `engine/docs/decisions/0010-rust-parameters-borrow-where-sound.md` + * for why that timing is what makes this analysis possible, where 0009 found it was not. The + * result is an input the shared lowerer and the Rust printer each consult; neither decides + * borrowing itself — see the `borrows` field on `LowerOptions` and the `borrowed`/`borrowedArgs` + * fields on the Target AST (`backend/tast.ts`). + * + * The search starts optimistic — every eligible parameter is assumed borrowable — and a parameter + * is demoted to owned the moment direct evidence shows it needs to be: it is returned, stored into + * a record field, a list element or a thrown error, assigned to, or forwarded unchanged to a + * parameter of another function that already needs to be owned. That last rule is a fixpoint over + * the call graph; `docs/semantics.md` §7 forbids recursion, so the call graph has no cycles and a + * worklist over its reverse edges always terminates (and would still terminate, just more slowly, + * if that ever changed — the worklist only ever adds each key once). + */ + +import type { CExpr, CFunc, CProgram, CStmt } from "../core/ir.ts"; +import type { SemType } from "../types.ts"; + +/** Qualified function name (`CFunc.name`) -> the names of its parameters that may be borrowed. */ +export type BorrowMap = ReadonlyMap>; + +/** + * Only a `String`, an `Enum` (also printed as Rust's owned `String` — `rustType`) or a `List` has + * an owned/borrowed distinction worth making in Rust. + */ +function isBorrowEligible(type: SemType): boolean { + return type.kind === "String" || type.kind === "Enum" || type.kind === "List"; +} + +/** "This identifier currently holds exactly this parameter's value, unchanged" — a pure alias. */ +type AliasScope = Map; + +/** `F.p` is forwarded, bare, into `G.q`: if `G.q` ends up owned, `F.p` must too (see the rule 4). */ +type Edge = { readonly fromKey: string; readonly toKey: string }; + +type WalkCtx = { + readonly program: CProgram; + readonly markOwned: (fnName: string, paramName: string) => void; + readonly edges: Edge[]; +}; + +function resolveAlias(expr: CExpr, scope: AliasScope): string | undefined { + return expr.kind === "local" ? scope.get(expr.name) : undefined; +} + +/** If `expr` is still a bare alias of one of `fn`'s own parameters, that parameter escapes here. */ +function markIfAlias(fn: CFunc, expr: CExpr, scope: AliasScope, ctx: WalkCtx): void { + const alias = resolveAlias(expr, scope); + if (alias !== undefined) ctx.markOwned(fn.name, alias); +} + +function walkBody(fn: CFunc, body: readonly CStmt[], scope: AliasScope, ctx: WalkCtx): void { + for (const statement of body) walkStmt(fn, statement, scope, ctx); +} + +function walkStmt(fn: CFunc, statement: CStmt, scope: AliasScope, ctx: WalkCtx): void { + switch (statement.kind) { + case "let": { + walkExpr(fn, statement.init, scope, ctx); + // A bare `let x = someParam;` extends the alias to `x`; anything else (a literal, a call, + // a concatenation) is a fresh, independent value from here on — whatever it needed from + // the parameter was already resolved (borrowed or cloned) at the point it was built, so + // the new local carries no further obligation back onto the parameter. + const alias = resolveAlias(statement.init, scope); + if (alias !== undefined) scope.set(statement.name, alias); + else scope.delete(statement.name); + return; + } + case "assign": + walkExpr(fn, statement.value, scope, ctx); + // Only a shadowing local can ever be an assignment target — `check.ts` marks every + // parameter binding `mutable: false`, so `assign` never targets one directly today. This + // check only fires if a future relaxation of that rule lets a name that still *is* the + // parameter (unshadowed in this scope) be reassigned, which a borrow could not survive. + if (scope.get(statement.name) === statement.name) ctx.markOwned(fn.name, statement.name); + return; + case "setIndex": + walkExpr(fn, statement.index, scope, ctx); + walkExpr(fn, statement.value, scope, ctx); + markIfAlias(fn, statement.value, scope, ctx); + return; + case "push": + walkExpr(fn, statement.value, scope, ctx); + markIfAlias(fn, statement.value, scope, ctx); + return; + case "if": + walkExpr(fn, statement.test, scope, ctx); + walkBody(fn, statement.then, new Map(scope), ctx); + walkBody(fn, statement.otherwise, new Map(scope), ctx); + return; + case "switch": + walkExpr(fn, statement.subject, scope, ctx); + for (const entry of statement.cases) walkBody(fn, entry.body, new Map(scope), ctx); + if (statement.otherwise !== undefined) walkBody(fn, statement.otherwise, new Map(scope), ctx); + return; + case "forRange": { + walkExpr(fn, statement.from, scope, ctx); + walkExpr(fn, statement.to, scope, ctx); + const inner = new Map(scope); + inner.delete(statement.name); + walkBody(fn, statement.body, inner, ctx); + return; + } + case "forEach": { + walkExpr(fn, statement.iterable, scope, ctx); + const inner = new Map(scope); + inner.delete(statement.name); + walkBody(fn, statement.body, inner, ctx); + return; + } + case "return": + if (statement.value !== undefined) { + walkExpr(fn, statement.value, scope, ctx); + markIfAlias(fn, statement.value, scope, ctx); + } + return; + case "fail": + for (const arg of statement.args) { + walkExpr(fn, arg, scope, ctx); + markIfAlias(fn, arg, scope, ctx); + } + return; + case "break": + case "continue": + return; + case "expr": + walkExpr(fn, statement.expr, scope, ctx); + return; + default: { + const exhaustive: never = statement; + return exhaustive; + } + } +} + +function walkExpr(fn: CFunc, expr: CExpr, scope: AliasScope, ctx: WalkCtx): void { + switch (expr.kind) { + case "lit": + case "none": + case "local": + return; + case "some": + walkExpr(fn, expr.inner, scope, ctx); + markIfAlias(fn, expr.inner, scope, ctx); + return; + case "record": + for (const field of expr.fields) { + walkExpr(fn, field.value, scope, ctx); + markIfAlias(fn, field.value, scope, ctx); + } + return; + case "list": + for (const item of expr.items) { + walkExpr(fn, item, scope, ctx); + markIfAlias(fn, item, scope, ctx); + } + return; + case "field": + walkExpr(fn, expr.target, scope, ctx); + return; + case "call": { + const callee = ctx.program.functions.get(expr.fn); + expr.args.forEach((arg, index) => { + walkExpr(fn, arg, scope, ctx); + const alias = resolveAlias(arg, scope); + if (alias === undefined) return; + const calleeParam = callee?.params[index]; + if (calleeParam === undefined) { + // A call this pass cannot resolve to a declared parameter (should not happen for + // the subset — every "call" targets a function in `program.functions`, and its + // arity matches) is the one place this pass cannot reduce "does the callee need + // this owned" to a fact, so it takes the safe side directly instead of guessing. + ctx.markOwned(fn.name, alias); + return; + } + ctx.edges.push({ fromKey: `${fn.name}::${alias}`, toKey: `${expr.fn}::${calleeParam.name}` }); + }); + return; + } + case "op": + // An intrinsic's own lowering decides its own borrowing, per target, independently of + // this map (the Rust backend's `borrowed()` — see 0010's "Positions this pass does not + // need to reach" section); this pass only has to keep looking for nested *calls* inside + // an operation's operands, not decide anything about the operation itself. + for (const arg of expr.args) walkExpr(fn, arg, scope, ctx); + return; + case "lambda": { + // A lambda is pure and, in every generated case, captures by reference rather than by + // move (`docs/decisions/0009-*.md`'s note on `task.race`), so reading an outer parameter + // inside one is just another read. Only a `return` inside the lambda body can hand a + // parameter's value somewhere that outlives the call — the list a combinator builds, or + // the `Option` a race closure produces — and `walkStmt`'s own "return" case already + // treats that the right way, so the lambda's body needs no special-casing beyond a scope + // of its own (its parameters shadow, and nothing declared inside it leaks back out). + const inner = new Map(scope); + for (const param of expr.params) inner.delete(param.name); + walkBody(fn, expr.body, inner, ctx); + return; + } + case "cond": + // Both branches always convert to an owned value at print time regardless of the source's + // borrow status (`toOwned`'s ternary case, in the Rust backend), so a branch that is a + // bare parameter alias never needs the parameter itself to be owned — it costs exactly + // the same clone either way. Still recurse, for a nested call in a branch. + walkExpr(fn, expr.test, scope, ctx); + walkExpr(fn, expr.then, scope, ctx); + walkExpr(fn, expr.otherwise, scope, ctx); + return; + case "and": + case "or": + walkExpr(fn, expr.left, scope, ctx); + walkExpr(fn, expr.right, scope, ctx); + return; + case "not": + walkExpr(fn, expr.operand, scope, ctx); + return; + default: { + const exhaustive: never = expr; + return exhaustive; + } + } +} + +export function computeBorrowableParams(program: CProgram): BorrowMap { + const owned = new Set(); + const edges: Edge[] = []; + const ctx: WalkCtx = { + program, + markOwned: (fnName, paramName) => owned.add(`${fnName}::${paramName}`), + edges, + }; + + for (const fn of program.functions.values()) { + const eligible = fn.params.filter((param) => isBorrowEligible(param.type)); + if (eligible.length === 0) continue; + const scope: AliasScope = new Map(eligible.map((param) => [param.name, param.name])); + walkBody(fn, fn.body, scope, ctx); + } + + // Reverse adjacency + a worklist seeded from the direct findings above: whenever a destination + // parameter ends up owned, every edge into it demotes its source too, and each key is only ever + // pushed once, so this is a standard (terminating) least-fixpoint-from-below computation. + const reverse = new Map(); + for (const edge of edges) { + const sources = reverse.get(edge.toKey); + if (sources === undefined) reverse.set(edge.toKey, [edge.fromKey]); + else sources.push(edge.fromKey); + } + const queue = [...owned]; + while (queue.length > 0) { + const key = queue.pop()!; + for (const source of reverse.get(key) ?? []) { + if (owned.has(source)) continue; + owned.add(source); + queue.push(source); + } + } + + const result = new Map>(); + for (const fn of program.functions.values()) { + const borrowable = new Set(); + for (const param of fn.params) { + if (isBorrowEligible(param.type) && !owned.has(`${fn.name}::${param.name}`)) borrowable.add(param.name); + } + result.set(fn.name, borrowable); + } + return result; +} diff --git a/engine/src/analysis/capabilities.ts b/engine/src/analysis/capabilities.ts new file mode 100644 index 000000000..b8bae8c7c --- /dev/null +++ b/engine/src/analysis/capabilities.ts @@ -0,0 +1,57 @@ +/** + * Capability threading. + * + * The checker already inferred each function's effects bottom-up over the call graph. This pass + * turns that into a decision every backend can print: which functions receive the capability + * record. Pure functions never do, which is what keeps generated code free of an ambient + * environment. + */ + +import { needsEnv, unionEffects } from "../effects.ts"; +import type { CFunc, CProgram } from "../core/ir.ts"; + +export function threadCapabilities(program: CProgram): CProgram { + const functions = new Map(program.functions); + + // Effects are already transitive, but a second closure keeps the pass independent of the order + // the checker happened to use, which matters once a frontend emits Core directly. + let changed = true; + while (changed) { + changed = false; + for (const [name, fn] of functions) { + const merged = unionEffects(fn.effects, ...fn.calls.map((callee) => functions.get(callee)?.effects ?? fn.effects)); + if ( + merged.fail.length !== fn.effects.fail.length || + merged.http !== fn.effects.http || + merged.clock !== fn.effects.clock || + merged.random !== fn.effects.random + ) { + functions.set(name, { ...fn, effects: merged }); + changed = true; + } + } + } + + for (const [name, fn] of functions) { + const usesEnv = needsEnv(fn.effects); + if (usesEnv !== fn.usesEnv) functions.set(name, { ...fn, usesEnv }); + } + + return { ...program, functions }; +} + +/** Every function reachable from the entry points, in dependency order. */ +export function dependencyClosure(program: CProgram, roots: readonly string[]): string[] { + const seen = new Set(); + const order: string[] = []; + const visit = (name: string): void => { + if (seen.has(name)) return; + seen.add(name); + const fn: CFunc | undefined = program.functions.get(name); + if (fn === undefined) return; + for (const callee of fn.calls) visit(callee); + order.push(name); + }; + for (const root of roots) visit(root); + return order; +} diff --git a/engine/src/api.ts b/engine/src/api.ts new file mode 100644 index 000000000..05887cbc8 --- /dev/null +++ b/engine/src/api.ts @@ -0,0 +1,48 @@ +/** + * The engine's public API. A host (the CLI, a test, another build tool) hands it a project and + * gets the annotated Core plus whatever targets were asked for. + */ + +import { Diagnostics } from "./diagnostics.ts"; +import { checkProgram } from "./core/check.ts"; +import type { CheckMetrics } from "./core/check.ts"; +import type { CProgram } from "./core/ir.ts"; +import { threadCapabilities } from "./analysis/capabilities.ts"; +import { optimize } from "./optimize/optimize.ts"; +import { link } from "./link/link.ts"; +import { loadModules } from "./project.ts"; +import type { ProjectConfig } from "./project.ts"; +import type { HModule } from "./hir/ast.ts"; + +export type CompileOptions = { + /** Skip the optimizer, so the Core dump matches the source one to one. */ + readonly noOptimize?: boolean; +}; + +export type Compilation = { + readonly modules: readonly HModule[]; + readonly program: CProgram; + readonly metrics: CheckMetrics; + readonly diagnostics: Diagnostics; +}; + +/** Parses, checks and analyses a project, stopping before any target is generated. */ +export function compileProject( + sourceRoot: string, + options: CompileOptions = {}, +): Compilation { + const diagnostics = new Diagnostics(); + const modules = loadModules(sourceRoot, diagnostics); + diagnostics.throwIfErrors(); + const { program, metrics } = checkProgram(modules, diagnostics); + diagnostics.throwIfErrors(); + const threaded = threadCapabilities(program); + const optimized = options.noOptimize === true ? threaded : optimize(threaded); + return { modules, program: link(optimized), metrics, diagnostics }; +} + +export type { ProjectConfig }; +export { loadConfig } from "./project.ts"; +export { dumpProgram } from "./core/ir.ts"; +export { dumpHir } from "./hir/ast.ts"; +export { renderDiagnostic, CompileError } from "./diagnostics.ts"; diff --git a/engine/src/backend/fold.ts b/engine/src/backend/fold.ts new file mode 100644 index 000000000..c79fd4a96 --- /dev/null +++ b/engine/src/backend/fold.ts @@ -0,0 +1,194 @@ +/** + * Constant folding over a lowered target AST. + * + * `optimize/optimize.ts` already folds the Core, and runs again after inlining so that a callee + * spliced into a call site sees its arguments as the constants they are. That is not the end of + * it: a lowering is itself code generation, and several of them expand an argument into a shape + * that only collapses once the argument is known. The TypeScript `date.fromYmd` expands a month + * into a days-in-month test, so `fromYmd(year, 3, day)` lowers to `3 === 2 ? 29 : 3 === 4 || … ? + * 30 : 31` — five comparisons and two branches that a reader can see are `31`, and that the Core + * folder never had a chance at because they did not exist when it ran. + * + * So the same fold runs once more here, on what each backend actually produced, before the + * printer sees it. It is deliberately narrower than the Core's: it folds comparisons between two + * literals, the boolean connectives, `!`, and a conditional or an `if` whose test is settled — + * nothing arithmetic, nothing target-specific, and nothing whose meaning differs between the four + * languages. Every operator spelling it recognizes (`===`/`==`, `<`, `and`, `&&`, …) means the + * same thing in every target that spells it that way, which is what makes one pass serve all of + * them. + */ + +import type { TExpr, TStmt } from "./tast.ts"; +import { mapExprs, mapStmts } from "./tast.ts"; + +/** Equality and ordering, in every spelling the four printers use. Value comparison in all of them. */ +const COMPARISONS = new Set(["===", "==", "!==", "!=", "<", "<=", ">", ">="]); +/** Conjunction and disjunction: `&&`/`||` in TypeScript, Go and Rust, `and`/`or` in Python. */ +const AND = new Set(["&&", "and"]); +const OR = new Set(["||", "or"]); +const NOT = new Set(["!", "not "]); + +type Primitive = boolean | number | bigint | string; + +/** The literal's value when it is one this pass is willing to reason about, else undefined. */ +function primitive(expr: TExpr): Primitive | undefined { + if (expr.kind !== "lit") return undefined; + const value = expr.value; + const kind = typeof value; + if (kind === "boolean" || kind === "number" || kind === "bigint" || kind === "string") { + return value as Primitive; + } + return undefined; +} + +function truth(expr: TExpr): boolean | undefined { + const value = primitive(expr); + return typeof value === "boolean" ? value : undefined; +} + +/** + * Compares two literals the way all four targets do, or answers undefined when they would not + * agree. A number and a string are never compared: JavaScript would coerce, Python would raise + * and Go and Rust would not compile, so a lowering that produced one is a bug to leave visible + * rather than a constant to fold. A number and a bigint are the same integer written two ways — + * the TypeScript target alone splits an `Int` between them by range (`needsBigInt`) — so those + * are compared as integers. + */ +function compare(op: string, left: Primitive, right: Primitive): boolean | undefined { + const numeric = (value: Primitive): bigint | undefined => { + if (typeof value === "bigint") return value; + if (typeof value === "number") return Number.isInteger(value) ? BigInt(value) : undefined; + return undefined; + }; + const leftNumber = numeric(left); + const rightNumber = numeric(right); + const comparable = + leftNumber !== undefined && rightNumber !== undefined + ? ([leftNumber, rightNumber] as const) + : typeof left === typeof right && (typeof left === "string" || typeof left === "boolean") + ? ([left, right] as [Primitive, Primitive]) + : undefined; + if (comparable === undefined) return undefined; + const [a, b] = comparable; + switch (op) { + case "===": + case "==": + return a === b; + case "!==": + case "!=": + return a !== b; + case "<": + return a < b; + case "<=": + return a <= b; + case ">": + return a > b; + case ">=": + return a >= b; + default: + return undefined; + } +} + +const TRUE: TExpr = { kind: "lit", value: true, type: { kind: "Bool" } }; +const FALSE: TExpr = { kind: "lit", value: false, type: { kind: "Bool" } }; + +/** Set by every fold that actually replaced something, so `foldTarget` knows when to stop. */ +type Progress = { changed: boolean }; + +function foldExpr(expr: TExpr, progress: Progress): TExpr { + switch (expr.kind) { + case "binary": { + if (COMPARISONS.has(expr.op)) { + const left = primitive(expr.left); + const right = primitive(expr.right); + if (left !== undefined && right !== undefined) { + const settled = compare(expr.op, left, right); + if (settled !== undefined) { + progress.changed = true; + return settled ? TRUE : FALSE; + } + } + return expr; + } + if (AND.has(expr.op)) { + const left = truth(expr.left); + if (left === false) { + progress.changed = true; + return FALSE; + } + if (left === true) { + progress.changed = true; + return expr.right; + } + // `x && false` is not `false`: `x` may be the call that does the work. Only a test + // whose *left* is settled lets the other side go. + return expr; + } + if (OR.has(expr.op)) { + const left = truth(expr.left); + if (left === true) { + progress.changed = true; + return TRUE; + } + if (left === false) { + progress.changed = true; + return expr.right; + } + return expr; + } + return expr; + } + case "unary": { + if (!NOT.has(expr.op)) return expr; + const operand = truth(expr.operand); + if (operand === undefined) return expr; + progress.changed = true; + return operand ? FALSE : TRUE; + } + case "ternary": { + const test = truth(expr.test); + if (test === undefined) return expr; + progress.changed = true; + return test ? expr.then : expr.otherwise; + } + default: + return expr; + } +} + +/** + * Drops the branch an `if` can no longer take. The surviving branch is spliced into the + * surrounding list rather than left nested, which is what the printer would otherwise emit as a + * block around statements that always run. + */ +function foldStmts(statements: readonly TStmt[], progress: Progress): TStmt[] { + return statements.flatMap((statement): TStmt[] => { + if (statement.kind !== "if") return [statement]; + const test = truth(statement.test); + if (test === undefined) return [statement]; + progress.changed = true; + return [...(test ? statement.then : statement.otherwise)]; + }); +} + +/** + * Folds until nothing more moves. One pass is not enough: collapsing a ternary can settle the + * test of the `if` that held it, and that `if` disappearing can expose another. The loop is + * bounded because every round that changes anything strictly removes a node. + */ +export function foldTarget(body: readonly TStmt[]): TStmt[] { + let current = [...body]; + for (let round = 0; round < 8; round++) { + const progress: Progress = { changed: false }; + current = foldStmts( + mapStmts( + mapExprs(current, (expr) => foldExpr(expr, progress)), + (statements) => foldStmts(statements, progress), + ), + progress, + ); + if (!progress.changed) break; + } + return current; +} diff --git a/engine/src/backend/format.ts b/engine/src/backend/format.ts new file mode 100644 index 000000000..853929cab --- /dev/null +++ b/engine/src/backend/format.ts @@ -0,0 +1,93 @@ +/** + * Running each target's own formatter over the generated output. + * + * The printer already emits the shape the code should have; a formatter only fixes whitespace, + * which is why a missing formatter never changes what the code means. It does change the bytes, + * though, and the generated output is committed, so a missing formatter is reported rather than + * passed over: `verify` refuses to run without all three, which is what keeps a checkout that + * lacks one from producing a diff against the committed files. + * + * Versions are pinned — prettier in the engine's own `package.json`, the others in + * `toolchain.lock.json` — so a formatter upgrade cannot silently rewrite a golden file. + */ + +import { spawnSync } from "node:child_process"; +import { existsSync } from "node:fs"; +import { resolve } from "node:path"; + +type Formatter = { + readonly name: string; + readonly command: string; + readonly args: readonly string[]; + /** How to ask the tool for its version, to tell "not installed" from "failed on this input". */ + readonly probe: readonly string[]; +}; + +/** + * Prettier from the engine's own dependencies, so the output does not depend on what the project + * being compiled happens to have installed. `npx` is the fallback for a checkout that has not run + * `npm install` yet. + */ +const PRETTIER = (): string => { + const local = resolve(import.meta.dirname, "..", "..", "node_modules", ".bin", "prettier"); + return existsSync(local) ? local : "npx"; +}; + +function prettierArgs(command: string): string[] { + const own = ["--write", "--use-tabs", "--no-config", "."]; + return command === "npx" ? ["--no-install", "prettier", ...own] : own; +} + +function formattersFor(target: string): Formatter[] { + switch (target) { + case "typescript": { + const command = PRETTIER(); + return [ + { + name: "prettier", + command, + args: prettierArgs(command), + probe: command === "npx" ? ["--no-install", "prettier", "--version"] : ["--version"], + }, + ]; + } + case "python": + return [{ name: "ruff format", command: "ruff", args: ["format", "--no-cache", "."], probe: ["--version"] }]; + case "go": + return [{ name: "gofmt", command: "gofmt", args: ["-w", "."], probe: ["-h"] }]; + case "rust": + // `cargo fmt` is rustfmt: it runs rustfmt over every file the crate's Cargo.toml lists, + // which a bare `rustfmt ` invocation would have to enumerate by hand. + return [{ name: "rustfmt", command: "cargo", args: ["fmt"], probe: ["fmt", "--version"] }]; + default: + return []; + } +} + +/** Whether a formatter is installed at all, as opposed to having failed on the input. */ +function isInstalled(formatter: Formatter): boolean { + const probe = spawnSync(formatter.command, [...formatter.probe], { encoding: "utf8", stdio: "pipe" }); + return probe.error === undefined && probe.status !== null; +} + +/** Formats a target's output in place, reporting both what ran and what was not installed. */ +export function formatOutput(target: string, outDir: string): { applied: string[]; missing: string[] } { + const applied: string[] = []; + const missing: string[] = []; + for (const formatter of formattersFor(target)) { + const result = spawnSync(formatter.command, [...formatter.args], { + cwd: outDir, + encoding: "utf8", + stdio: "pipe", + }); + if (result.status === 0) applied.push(formatter.name); + else if (!isInstalled(formatter)) missing.push(formatter.name); + else applied.push(`${formatter.name} (failed)`); + } + return { applied, missing }; +} + +/** The formatters a target needs, whether or not they are installed. Used by `verify`. */ +export function formattersOf(target: string): { name: string; installed: boolean }[] { + return formattersFor(target).map((formatter) => ({ name: formatter.name, installed: isInstalled(formatter) })); +} diff --git a/engine/src/backend/generate.ts b/engine/src/backend/generate.ts new file mode 100644 index 000000000..d67c85090 --- /dev/null +++ b/engine/src/backend/generate.ts @@ -0,0 +1,557 @@ +/** + * Generation: Target AST to files on disk. + * + * Everything here is target-independent. A backend contributes a spec (naming, passes), a printer + * and its support module; this module decides the file layout, the imports between generated + * modules, the provenance header, and the three review artifacts every target produces: + * `LOWERING.md`, `API.json` and `SOURCEMAP.json`. + */ + +import { createHash } from "node:crypto"; +import { existsSync, mkdirSync, readFileSync, readdirSync, rmSync, statSync, writeFileSync } from "node:fs"; +import { dirname, join, relative } from "node:path"; +import type { CExpr, CProgram, CStmt } from "../core/ir.ts"; +import type { SemType } from "../types.ts"; +import { typeToString } from "../types.ts"; +import { lowerProgram } from "./lower.ts"; +import type { LowerOptions, TargetSpec } from "./lower.ts"; +import type { TExpr, TFunc, TImport, TModule } from "./tast.ts"; +import { inlineCalls } from "../optimize/inline.ts"; +import type { InlineBudget } from "../optimize/inline.ts"; +import { optimize } from "../optimize/optimize.ts"; +import { foldTarget } from "./fold.ts"; + +export const ENGINE_VERSION = "0.1.0"; + +export type Backend = { + readonly spec: TargetSpec; + readonly fileExtension: string; + /** Renders a module. Imports have already been computed. */ + readonly printModule: (module: TModule) => string; + /** + * Extra `LowerOptions` this backend needs computed from the whole program, merged in before + * `lowerProgram` runs. This is how the Rust backend hands its whole-program borrow pre-pass + * (`analysis/borrows.ts`) to the shared lowerer without `generate` — target-independent by + * design — branching on which target it is building: every other backend simply has none. + */ + readonly extraLowerOptions?: (program: CProgram) => Partial; + /** How module `from` refers to module `to` in an import. */ + readonly importPath: (from: string, to: string) => string; + /** The generated capability and concurrency support, when the program needs it. */ + readonly support?: (program: CProgram, needs: SupportNeeds) => { path: string; text: string } | undefined; + /** The generated declarations of the project's domain errors. */ + readonly errorsModule?: (program: CProgram) => { path: string; text: string } | undefined; + /** A type rendered in `API.json`, so DX authors see the real signature. */ + readonly renderType: (type: SemType) => string; + /** What a module imports from the target's support file, when it needs one. */ + readonly supportImport?: ( + needs: SupportNeeds, + usesEnv: boolean, + builtins: readonly string[], + ) => readonly { from: string; names: readonly string[]; typeOnly?: boolean }[]; + /** Line comment marker, used for the provenance header. */ + readonly comment?: string; + /** + * A program that reads `{"fn": …, "args": […]}` lines and answers `{"ok": …}` lines, so the + * differential harness can drive every target through one protocol. + */ + readonly driver?: (program: CProgram, entryPoints: readonly DriverEntry[]) => { path: string; text: string }[]; + /** + * How this target reaches its own platform default capabilities, present only when it has one + * to reach. TypeScript and Python do; Go and Rust ship no concrete implementation in the + * generated core at all — see `docs/decisions/0011-public-entry-points-vs-capabilities.md`. + * When set, `generate` gives every capability-taking entry point a public wrapper under the + * source's own name and no capability parameter, moving the capability-taking implementation + * to an internal seam this expression feeds by default. Absent, the capability-taking form + * stays the only entry point, marked in `API.json` rather than hidden behind a fake. + */ + readonly defaultCapabilities?: { + /** + * A bare reference to a module-level singleton, built once at load time — never a call + * repeated per invocation, which is the whole point of building it once. + */ + readonly ref: TExpr; + /** What a module with a wrapper needs to import to see `ref`. */ + readonly imports: readonly { from: string; names: readonly string[] }[]; + /** The seam's own name, derived from the public wrapper's name it stands in for. */ + readonly seamName: (publicName: string) => string; + }; + /** + * How aggressively this target's own call-site inlining (`optimize/inline.ts`) should run, + * absent when the target's own compiler already does this job — see that module's header for + * why the budget is per target rather than a single number for every backend. + */ + readonly inlineBudget?: InlineBudget; +}; + +/** One entry point, as the generated driver sees it. */ +export type DriverEntry = { + readonly coreName: string; + readonly targetName: string; + readonly modulePath: string; + readonly params: readonly SemType[]; + readonly ret: SemType; + readonly usesEnv: boolean; + readonly fails: readonly string[]; +}; + +export type SupportNeeds = { + readonly env: boolean; + readonly race: boolean; +}; + +export type GenerateResult = { + readonly files: readonly { readonly path: string; readonly text: string }[]; + readonly lowering: string; + /** Every selection the lowering table made, for the metrics. */ + readonly selections: readonly { op: string; args: string; impl: string; reason: string }[]; + readonly api: unknown; + readonly sourceMap: unknown; +}; + +export function generate( + program: CProgram, + backend: Backend, + options: LowerOptions = {}, +): GenerateResult { + // Inlining substitutes a call site's own arguments into the callee's body, so it exposes + // constants the folder could not see the first time it ran: a helper that takes a month and + // asks whether it is February, called with `3`, becomes `3 === 2` — dead, but only once the + // argument is in place. Re-running the Core-to-Core passes afterwards is what turns that into + // the branch a person would have written. They are the same passes, proven against the + // reference interpreter by `tests/translation.spec.ts`, and running them twice is idempotent + // where there is nothing new to find. + const inlined = + backend.inlineBudget === undefined ? program : optimize(inlineCalls(program, backend.inlineBudget)); + const lowered = lowerProgram(inlined, backend.spec, { ...options, ...backend.extraLowerOptions?.(inlined) }); + const moduleOf = new Map(); + for (const fn of inlined.functions.values()) moduleOf.set(fn.name, fn.module); + + // After lowering, because a lowering is code generation too: the shape it expands an argument + // into only collapses once that argument is a constant, which inlining is what makes it. See + // `fold.ts` for what this is allowed to fold and why one pass serves all four targets. + const modules = splitCapabilityEntryPoints(lowered.modules, backend).map((module) => ({ + ...module, + functions: module.functions.map((fn) => ({ ...fn, body: foldTarget(fn.body) })), + })); + + const files: { path: string; text: string }[] = []; + const sourceMap: Record = {}; + const api: { + functions: { name: string; module: string; params: { name: string; type: string }[]; returns: string; effects: string[] }[]; + // A capability-taking form: either an entry point's internal seam (its public wrapper is + // listed in `functions` instead) or, absent a wrapper, the same entry as `functions` lists, + // repeated here so a reader sees it needs capabilities without inferring that from `effects`. + seams: { + name: string; + publicName: string; + module: string; + params: { name: string; type: string }[]; + returns: string; + hasWrapper: boolean; + }[]; + records: { name: string; fields: { name: string; type: string }[] }[]; + errors: string[]; + } = { functions: [], seams: [], records: [], errors: [] }; + + const needs: SupportNeeds = { + env: [...inlined.functions.values()].some((fn) => fn.usesEnv), + race: [...inlined.functions.values()].some((fn) => usesOp(fn.body, "task.race")), + }; + + for (const module of modules) { + const sourcePath = module.sourcePath; + const imports = computeImports( + module, + sourcePath, + inlined, + lowered.functionNames, + moduleOf, + backend, + needs, + lowered.moduleNeeds.get(sourcePath) ?? new Set(), + [...(lowered.moduleBuiltins.get(sourcePath) ?? new Set())].sort(), + ); + const hasWrapper = module.functions.some((fn) => fn.seamName !== undefined); + if (hasWrapper && backend.defaultCapabilities !== undefined) { + for (const item of backend.defaultCapabilities.imports) { + // Folded into an existing import from the same place, typed the same way, rather than + // a second statement — `supportImport` already returns one untyped import per source + // for a target that only ever needs one, and this keeps that target at just the one. + const index = imports.findIndex( + (candidate) => candidate.from === item.from && candidate.typeOnly !== true, + ); + if (index === -1) { + imports.push({ from: item.from, names: [...item.names] }); + } else { + const merged = [...new Set([...imports[index]!.names, ...item.names])].sort(); + imports[index] = { ...imports[index]!, names: merged }; + } + } + } + const printed = backend.printModule({ + ...module, + imports, + header: header(sourcePath, program, backend.comment ?? "//"), + }); + files.push({ path: module.path, text: printed }); + + for (const fn of module.functions) { + sourceMap[`${module.path}#${fn.name}`] = { + module: fn.source.module, + start: fn.source.start, + end: fn.source.end, + }; + if (fn.exported) { + api.functions.push({ + name: fn.name, + module: module.path, + params: fn.params.map((param) => ({ name: param.name, type: backend.renderType(param.type) })), + returns: backend.renderType(fn.ret), + effects: fn.fails.map((name) => `Fail<${name}>`).concat(fn.usesEnv ? ["env"] : []), + }); + } + if (fn.seam === true) { + api.seams.push({ + name: fn.name, + publicName: fn.wrapperName ?? fn.name, // no wrapper: it is its own public name + module: module.path, + params: fn.params.map((param) => ({ name: param.name, type: backend.renderType(param.type) })), + returns: backend.renderType(fn.ret), + hasWrapper: fn.wrapperName !== undefined, + }); + } + } + for (const record of module.records) { + if (api.records.some((item) => item.name === record.name)) continue; + api.records.push({ + name: record.name, + fields: record.fields.map((field) => ({ name: field.name, type: backend.renderType(field.type) })), + }); + } + } + + const entries: DriverEntry[] = []; + for (const module of modules) { + for (const fn of module.functions) { + // The wrapper itself never drives: it has no capability parameter to inject a fake + // through, so the differential harness always targets its seam instead (below). + if (fn.seamName !== undefined) continue; + if (!fn.exported && fn.seam !== true) continue; + entries.push({ + coreName: `${module.sourcePath}::${fn.source.name}`, + targetName: fn.name, + modulePath: module.path, + params: fn.params.filter((param) => param.name !== "env").map((param) => param.type), + ret: fn.ret, + usesEnv: fn.usesEnv, + fails: fn.fails, + }); + } + } + for (const file of backend.driver?.(program, entries) ?? []) files.push(file); + + const errors = backend.errorsModule?.(program); + if (errors !== undefined) files.push(errors); + const support = backend.support?.(program, needs); + if (support !== undefined) files.push(support); + api.errors = [...program.errors.keys()].sort(); + + return { + files, + lowering: lowered.table.renderLoweringDoc(backend.spec.name), + selections: lowered.table.selections, + api, + sourceMap, + }; +} + +/** + * Defect 1's fix (`docs/decisions/0011-public-entry-points-vs-capabilities.md`): capability + * threading gives an entry point an `env` parameter the moment it or something it calls reaches + * Http, Clock or Random, but the source never declared that parameter, so it cannot stay on the + * function a caller imports under the entry point's own name. Where the target can build a + * default (`backend.defaultCapabilities`), this splits such an entry point in two: a public + * wrapper that keeps the source's exact signature and calls an internal seam — named so it reads + * as one — that still takes capabilities and defaults to the platform's own. The seam is what the + * differential driver calls directly, to inject fakes (`entries`, below). Where the target cannot + * (Go, Rust — see the module comment on `defaultCapabilities`), the function is left as is, only + * marked (`seam: true`) so `API.json` documents it as capability-taking rather than an ordinary + * utility. + */ +function splitCapabilityEntryPoints(modules: readonly TModule[], backend: Backend): TModule[] { + const defaults = backend.defaultCapabilities; + return modules.map((module) => { + let changed = false; + const functions = module.functions.flatMap((fn): TFunc[] => { + if (!fn.exported || !fn.usesEnv) return [fn]; + changed = true; + if (defaults === undefined) return [{ ...fn, seam: true }]; + + const seamName = defaults.seamName(fn.name); + const publicParams = fn.params.filter((param) => param.name !== "env"); + const wrapper: TFunc = { + name: fn.name, + params: publicParams, + ret: fn.ret, + body: [ + { + kind: "return", + value: { + kind: "call", + callee: { kind: "name", name: seamName }, + args: [...publicParams.map((param): TExpr => ({ kind: "name", name: param.name })), defaults.ref], + await: fn.isAsync, + }, + }, + ], + exported: true, + moduleExported: true, + doc: fn.doc, + isAsync: fn.isAsync, + fails: fn.fails, + usesEnv: false, + seamName, + source: fn.source, + }; + // The seam carries its own short note rather than a copy of the utility's documentation: + // a reader who reaches it is looking for why it exists, and the utility's own prose is + // already right above it on the wrapper. + const seamNote = + `\`${fn.name}\`, taking its capabilities explicitly.\n\n` + + `The public \`${fn.name}\` calls this with the platform's defaults. Pass your own to\n` + + "supply a clock, a source of randomness or an HTTP client — which is what the\n" + + "differential conformance driver does to make a run reproducible."; + const seam: TFunc = { + ...fn, + name: seamName, + exported: false, + moduleExported: true, + seam: true, + wrapperName: fn.name, + doc: seamNote, + }; + return [wrapper, seam]; + }); + return changed ? { ...module, functions } : module; + }); +} + +/** Whether a Core body mentions an operation, used to decide what support code is needed. */ +function usesOp(body: readonly CStmt[], op: string): boolean { + let found = false; + const expr = (node: CExpr): void => { + if (found) return; + switch (node.kind) { + case "op": + if (node.op === op) found = true; + node.args.forEach(expr); + return; + case "call": + node.args.forEach(expr); + return; + case "record": + node.fields.forEach((field) => expr(field.value)); + return; + case "list": + node.items.forEach(expr); + return; + case "field": + expr(node.target); + return; + case "some": + expr(node.inner); + return; + case "cond": + expr(node.test); + expr(node.then); + expr(node.otherwise); + return; + case "and": + case "or": + expr(node.left); + expr(node.right); + return; + case "not": + expr(node.operand); + return; + case "lambda": + node.body.forEach(statement); + return; + default: + return; + } + }; + const statement = (node: CStmt): void => { + switch (node.kind) { + case "let": + expr(node.init); + return; + case "assign": + case "push": + expr(node.value); + return; + case "setIndex": + expr(node.index); + expr(node.value); + return; + case "if": + expr(node.test); + node.then.forEach(statement); + node.otherwise.forEach(statement); + return; + case "switch": + expr(node.subject); + node.cases.forEach((entry) => entry.body.forEach(statement)); + node.otherwise?.forEach(statement); + return; + case "forRange": + expr(node.from); + expr(node.to); + node.body.forEach(statement); + return; + case "forEach": + expr(node.iterable); + node.body.forEach(statement); + return; + case "return": + if (node.value !== undefined) expr(node.value); + return; + case "fail": + node.args.forEach(expr); + return; + case "expr": + expr(node.expr); + return; + default: + return; + } + }; + body.forEach(statement); + return found; +} + +function header(modulePath: string, program: CProgram, comment: string): string { + const hash = createHash("sha256") + .update( + [...program.functions.values()] + .filter((fn) => fn.module === modulePath) + .map((fn) => fn.name) + .sort() + .join(","), + ) + .digest("hex") + .slice(0, 12); + return [ + `${comment} Code generated by the logic engine. DO NOT EDIT.`, + `${comment} engine: ${ENGINE_VERSION}`, + `${comment} source: ${modulePath}`, + `${comment} content: ${hash}`, + ].join("\n"); +} + +function computeImports( + module: TModule, + sourcePath: string, + program: CProgram, + names: ReadonlyMap, + moduleOf: ReadonlyMap, + backend: Backend, + needs: SupportNeeds, + portable: ReadonlySet, + builtins: readonly string[], +): TImport[] { + const byModule = new Map>(); + const record = (callee: string): void => { + const target = moduleOf.get(callee); + if (target === undefined || target === sourcePath) return; + const name = names.get(callee); + if (name === undefined) return; + const set = byModule.get(target); + if (set === undefined) byModule.set(target, new Set([name])); + else set.add(name); + }; + for (const callee of portable) record(callee); + for (const fn of program.functions.values()) { + if (fn.module !== sourcePath) continue; + for (const callee of fn.calls) { + record(callee); + } + } + const imports: TImport[] = [...byModule.entries()] + .sort(([left], [right]) => left.localeCompare(right)) + .map(([target, used]) => ({ + from: backend.importPath(sourcePath, target), + names: [...used].sort(), + })); + + const localErrors = [...program.errors.values()].filter((error) => !error.exported || true); + const usesErrors = module.functions.some((fn) => fn.fails.length > 0); + if (usesErrors && localErrors.length > 0) { + imports.push({ + from: backend.importPath(sourcePath, "errors"), + names: [...new Set(module.functions.flatMap((fn) => fn.fails))].sort(), + }); + } + const usesEnv = module.functions.some((fn) => fn.usesEnv); + for (const support of backend.supportImport?.(needs, usesEnv, builtins) ?? []) { + if (support.names.length > 0) { + imports.push({ from: support.from, names: [...support.names], typeOnly: support.typeOnly }); + } + } + return imports; +} + +/** Directories that belong to the target's own toolchain, never to this engine. */ +const FOREIGN_DIRECTORIES = new Set(["target", "__pycache__", "node_modules", ".git"]); + +/** + * Removes generated files this run did not produce. + * + * A module stops being generated whenever the program stops needing it — inlining a helper into + * its only caller drops the whole file from the dependency closure — and a leftover from an + * earlier run is not inert: every later step reads the directory rather than the result, so a + * stale file is typechecked, benchmarked, measured and committed as though it were output. It is + * identified by this engine's own header and by nothing else, so a fixture, a lockfile or a + * toolchain's build directory sitting beside the output is never touched. + */ +function removeStaleFiles(outDir: string, written: ReadonlySet): void { + if (!existsSync(outDir)) return; + const walk = (directory: string): void => { + for (const entry of readdirSync(directory)) { + const full = join(directory, entry); + if (statSync(full).isDirectory()) { + if (!FOREIGN_DIRECTORIES.has(entry)) walk(full); + continue; + } + if (written.has(relative(outDir, full))) continue; + let head: string; + try { + head = readFileSync(full, "utf8").slice(0, 200); + } catch { + continue; + } + if (head.includes("Code generated by the logic engine")) rmSync(full); + } + }; + walk(outDir); +} + +export function writeFiles(outDir: string, result: GenerateResult, extras: Record): void { + const written = new Set(); + for (const file of [...result.files]) { + const full = join(outDir, file.path); + mkdirSync(dirname(full), { recursive: true }); + writeFileSync(full, file.text); + written.add(relative(outDir, full)); + } + for (const [name, text] of Object.entries(extras)) { + const full = join(outDir, name); + mkdirSync(dirname(full), { recursive: true }); + writeFileSync(full, text); + written.add(relative(outDir, full)); + } + removeStaleFiles(outDir, written); +} + +export { relative, typeToString }; diff --git a/engine/src/backend/lower.ts b/engine/src/backend/lower.ts new file mode 100644 index 000000000..e87741241 --- /dev/null +++ b/engine/src/backend/lower.ts @@ -0,0 +1,754 @@ +/** + * Core to Target AST. + * + * This is the one place that knows how a semantic operation becomes target structure; what each + * operation looks like is decided by the target's capability table (lowering selection), and how + * it is printed is decided by the target's printer. The passes a target needs — hoisting + * expressions into statements, turning combinators into loops, turning `Fail` into a second + * return value, colouring async — are parameterized here rather than duplicated per backend. + */ + +import { dependencyClosure } from "../analysis/capabilities.ts"; +import type { BorrowMap } from "../analysis/borrows.ts"; +import type { CExpr, CFunc, CProgram, CStmt } from "../core/ir.ts"; +import type { NormalizedRegex } from "../regex.ts"; +import type { SemType } from "../types.ts"; +import { tBool, tString } from "../types.ts"; +import { BUILTIN_RECORDS } from "../intrinsics/index.ts"; +import { LoweringTable } from "./select.ts"; +import type { EmitContext } from "./select.ts"; +import { mapExprs } from "./tast.ts"; +import type { TExpr, TFunc, TModule, TParam, TRecord, TStmt } from "./tast.ts"; + +export type TargetSpec = { + readonly name: string; + readonly table: LoweringTable; + /** Identifier style for functions, parameters, locals and record fields. */ + readonly naming: { + readonly func: (localName: string, exported: boolean) => string; + readonly value: (name: string) => string; + readonly field: (name: string, exported: boolean) => string; + readonly type: (name: string) => string; + readonly module: (path: string) => string; + }; + /** Operations this target prefers to print as a statement-level loop rather than a call. */ + readonly loopCombinators: ReadonlySet; + /** True when a fallible function returns `(value, error)` instead of raising. */ + readonly errorsAsValues: boolean; + /** True when functions that reach Http become async and their calls are awaited. */ + readonly asyncColouring: boolean; + /** True when the target has no conditional expression, so `cond` is hoisted into statements. */ + readonly statementTernary: boolean; + /** Extra modules the generated file must import for a given capability. */ + readonly envType: SemType; +}; + +export type LowerOptions = { + /** Emit the plain combinator-to-loop form instead of target idioms. */ + readonly noIdioms?: boolean; + /** + * Which String/List parameters may be printed as borrows rather than owned values (the Rust + * backend's own pre-pass — see `analysis/borrows.ts`). The lowerer only consults this map, on + * `TParam.borrowed` and on a "call" node's `borrowedArgs`; it never decides borrowing itself. + * Absent for every target that has no such distinction, which is every target but Rust. + */ + readonly borrows?: BorrowMap; +}; + +export type LoweredProgram = { + readonly modules: readonly TModule[]; + readonly table: LoweringTable; + readonly functionNames: ReadonlyMap; + /** Core functions each generated module calls through a portable lowering, so imports are complete. */ + readonly moduleNeeds: ReadonlyMap>; + /** Engine-defined record types each module uses, which live in the target's support file. */ + readonly moduleBuiltins: ReadonlyMap>; +}; + +const ENV_PARAM = "env"; + +const BUILTIN_RECORD_NAMES: readonly string[] = BUILTIN_RECORDS.map((record) => record.name); + +export function lowerProgram( + program: CProgram, + spec: TargetSpec, + options: LowerOptions = {}, +): LoweredProgram { + return new Lowerer(program, spec, options).run(); +} + +class Lowerer { + private readonly program: CProgram; + private readonly spec: TargetSpec; + private readonly options: LowerOptions; + private readonly names = new Map(); + private readonly imports = new Map>(); + private readonly needed = new Set(); + private readonly moduleNeeds = new Map>(); + private readonly moduleBuiltins = new Map>(); + private pending: TStmt[] = []; + private constants: { name: string; type: SemType; value: TExpr }[] = []; + private readonly constantNames = new Map(); + private temporaries = 0; + private currentModule = ""; + private currentFails: readonly string[] = []; + private currentReturn: SemType = tBool; + /** + * The (Core-named, not yet target-renamed) parameters of the function currently being lowered + * that `options.borrows` found borrowable. Read by `expr`'s "local" case so a reference to one + * of them carries `borrowed: true` from the moment it is built — including from inside a + * candidate's own `emit`, which runs here, before the function it belongs to has a printed + * form at all. That timing is exactly what `docs/decisions/0010-*.md` relies on: this field is + * set once, before a function's body is lowered, not discovered from print-time scope. + */ + private currentBorrowedParams: ReadonlySet = new Set(); + /** + * Functions some *other* generated module calls. A source module's own `export` decides a + * function's visibility (`CFunc.moduleExported`), but a specialization (ADR 0004) is not the + * declaration it came from and carries that flag as false — which is right until the + * specialization survives as its own function and is called across a module boundary, where + * the importing file would name something the defining file never made visible. Whatever one + * generated module imports, the module that defines it exports. + */ + private crossModuleCallees: ReadonlySet = new Set(); + + constructor(program: CProgram, spec: TargetSpec, options: LowerOptions) { + this.program = program; + this.spec = spec; + this.options = options; + } + + run(): LoweredProgram { + this.crossModuleCallees = this.findCrossModuleCallees(); + // Two passes: the first discovers which source library functions the selected portable + // lowerings need, the second lowers with those functions in the closure and named. + this.lowerAll(this.closure(this.program.entryPoints)); + const roots = [...this.program.entryPoints, ...this.needed].filter((name) => + this.program.functions.has(name), + ); + this.moduleNeeds.clear(); + this.moduleBuiltins.clear(); + return { + modules: this.lowerAll(this.closure(roots)), + table: this.spec.table, + functionNames: this.names, + moduleNeeds: this.moduleNeeds, + moduleBuiltins: this.moduleBuiltins, + }; + } + + private findCrossModuleCallees(): ReadonlySet { + const found = new Set(); + for (const caller of this.program.functions.values()) { + for (const callee of caller.calls) { + const target = this.program.functions.get(callee); + if (target !== undefined && target.module !== caller.module) found.add(callee); + } + } + return found; + } + + private closure(roots: readonly string[]): CFunc[] { + return dependencyClosure(this.program, roots) + .map((name) => this.program.functions.get(name)) + .filter((fn): fn is CFunc => fn !== undefined); + } + + private lowerAll(functions: readonly CFunc[]): TModule[] { + this.assignNames(functions); + + const byModule = new Map(); + for (const fn of functions) { + const list = byModule.get(fn.module); + if (list === undefined) byModule.set(fn.module, [fn]); + else list.push(fn); + } + + const modules: TModule[] = []; + for (const [modulePath, moduleFunctions] of byModule) { + this.currentModule = modulePath; + this.imports.set(modulePath, new Set()); + const records = this.recordsOf(moduleFunctions); + this.constants = []; + this.constantNames.clear(); + const lowered = moduleFunctions + .map((fn) => this.lowerFunction(fn)) + .map((fn) => ({ ...fn, body: this.hoistConstantTables(fn.body) })); + modules.push({ + path: this.spec.naming.module(modulePath), + sourcePath: modulePath, + imports: [], + records, + errors: [], + constants: this.constants, + functions: lowered, + header: "", + requires: [...(this.imports.get(modulePath) ?? [])].sort(), + }); + } + return modules; + } + + private assignNames(functions: readonly CFunc[]): void { + this.names.clear(); + const used = new Map(); + for (const fn of functions) { + const base = this.spec.naming.func(fn.localName.replaceAll("$", "_"), fn.exported); + const count = used.get(base) ?? 0; + used.set(base, count + 1); + this.names.set(fn.name, count === 0 ? base : `${base}_${count}`); + } + } + + /** Record types the engine itself defines; each target declares them once in its support file. */ + private builtinRecordsUsed = new Set(); + + private recordsOf(functions: readonly CFunc[]): TRecord[] { + const used = new Set(); + const collectType = (type: SemType): void => { + switch (type.kind) { + case "Record": + used.add(type.name); + break; + case "List": + collectType(type.elem); + break; + case "Option": + collectType(type.inner); + break; + default: + break; + } + }; + for (const fn of functions) { + for (const param of fn.params) collectType(param.type); + collectType(fn.ret); + } + const records: TRecord[] = []; + for (const name of used) { + const definition = this.program.records.get(name); + if (definition === undefined) continue; + if (BUILTIN_RECORD_NAMES.includes(name)) { + this.builtinRecordsUsed.add(name); + const perModule = this.moduleBuiltins.get(this.currentModule) ?? new Set(); + perModule.add(name); + this.moduleBuiltins.set(this.currentModule, perModule); + continue; + } + records.push({ + name: this.spec.naming.type(name), + doc: definition.doc, + fields: definition.fields.map((field) => ({ + name: this.spec.naming.field(field.name, true), + type: field.optional && field.type.kind !== "Option" ? { kind: "Option", inner: field.type } : field.type, + doc: field.doc, + })), + }); + } + return records; + } + + /* ---------------------------------------------------------------- * + * Functions + * ---------------------------------------------------------------- */ + + private lowerFunction(fn: CFunc): TFunc { + this.temporaries = 0; + this.currentFails = fn.effects.fail; + this.currentReturn = fn.ret; + const borrowedHere = this.options.borrows?.get(fn.name) ?? new Set(); + const previousBorrowed = this.currentBorrowedParams; + this.currentBorrowedParams = borrowedHere; + const params: TParam[] = fn.params.map((param) => ({ + name: this.spec.naming.value(param.name), + type: param.type, + doc: param.doc, + borrowed: borrowedHere.has(param.name), + })); + if (fn.usesEnv) params.push({ name: ENV_PARAM, type: this.spec.envType }); + try { + return { + name: this.names.get(fn.name) ?? fn.localName, + params, + ret: fn.ret, + body: this.block(fn.body), + exported: fn.exported, + moduleExported: fn.moduleExported || this.crossModuleCallees.has(fn.name), + doc: fn.doc, + isAsync: this.spec.asyncColouring && fn.effects.http, + fails: fn.effects.fail, + usesEnv: fn.usesEnv, + source: { module: fn.module, name: fn.localName, start: fn.span.start, end: fn.span.end }, + }; + } finally { + this.currentBorrowedParams = previousBorrowed; + } + } + + private block(body: readonly CStmt[]): TStmt[] { + const outer = this.pending; + const result: TStmt[] = []; + for (const statement of body) { + this.pending = []; + const lowered = this.statement(statement); + result.push(...this.pending, ...lowered); + } + this.pending = outer; + return result; + } + + private statement(statement: CStmt): TStmt[] { + switch (statement.kind) { + case "let": { + const loop = this.combinatorLoop(statement.init, statement.name, statement.type); + if (loop !== undefined) return loop; + return [ + { + kind: "let", + name: this.spec.naming.value(statement.name), + type: statement.type, + init: this.expr(statement.init), + mutable: statement.mutable, + }, + ]; + } + case "assign": + return [ + { + kind: "assign", + target: { kind: "name", name: this.spec.naming.value(statement.name) }, + value: this.expr(statement.value), + }, + ]; + case "setIndex": + return [ + { + kind: "assign", + target: { + kind: "index", + target: { kind: "name", name: this.spec.naming.value(statement.name) }, + index: this.expr(statement.index), + }, + value: this.expr(statement.value), + }, + ]; + case "push": + return [this.pushStatement(statement)]; + case "if": { + const test = this.expr(statement.test); + return [ + { + kind: "if", + test, + then: this.block(statement.then), + otherwise: this.block(statement.otherwise), + }, + ]; + } + case "switch": + return [ + { + kind: "switch", + subject: this.expr(statement.subject), + cases: statement.cases.map((entry) => ({ + values: entry.values, + body: this.block(entry.body), + })), + otherwise: statement.otherwise === undefined ? undefined : this.block(statement.otherwise), + }, + ]; + case "forRange": + return [ + { + kind: "for", + name: this.spec.naming.value(statement.name), + type: statement.type, + from: this.expr(statement.from), + to: this.expr(statement.to), + inclusive: statement.inclusive, + step: statement.step, + body: this.block(statement.body), + }, + ]; + case "forEach": + return [ + { + kind: "forEach", + name: this.spec.naming.value(statement.name), + type: statement.type, + iterable: this.expr(statement.iterable), + body: this.block(statement.body), + }, + ]; + case "return": { + if (statement.value === undefined) return [{ kind: "return" }]; + const value = this.expr(statement.value); + return [ + this.spec.errorsAsValues && this.currentFails.length > 0 + ? { kind: "return", value, extra: [{ kind: "raw", text: "nil" }] } + : { kind: "return", value }, + ]; + } + case "fail": + return [ + { + kind: "throw", + errorClass: this.spec.naming.type(statement.errorClass), + args: statement.args.map((arg) => this.expr(arg)), + }, + ]; + case "break": + return [{ kind: "break" }]; + case "continue": + return [{ kind: "continue" }]; + case "expr": + return [{ kind: "expr", expr: this.expr(statement.expr) }]; + default: { + const exhaustive: never = statement; + return exhaustive; + } + } + } + + private pushStatement(statement: Extract): TStmt { + const list: TExpr = { kind: "name", name: this.spec.naming.value(statement.name) }; + const value = this.expr(statement.value); + const selection = this.spec.table.select("seq.push", []); + return { kind: "expr", expr: selection.candidate.emit([list, value], [], this.context()) }; + } + + /** + * Turns `const total = seq.fold(xs, 0, f)` into a loop when the target prefers one. This is + * the `combinator-to-loop` nanopass, and `--no-idioms` forces it on for every combinator so + * conformance can compare the idiomatic and the plain form. + */ + private combinatorLoop(init: CExpr, name: string, type: SemType): TStmt[] | undefined { + if (init.kind !== "op") return undefined; + const wanted = this.options.noIdioms === true || this.spec.loopCombinators.has(init.op); + if (!wanted) return undefined; + const target = this.spec.naming.value(name); + + if (init.op === "seq.fold") { + const [list, initial, fn] = init.args; + if (fn === undefined || fn.kind !== "lambda" || list === undefined || initial === undefined) { + return undefined; + } + const element = this.spec.naming.value(fn.params[1]!.name); + const accumulator = fn.params[0]!.name; + const body = this.inlineLambdaBody(fn, [ + { from: accumulator, to: target }, + ]); + return [ + { kind: "let", name: target, type, init: this.expr(initial), mutable: true }, + { + kind: "forEach", + name: element, + type: fn.params[1]!.type, + iterable: this.expr(list), + body: body.map((statement) => + statement.kind === "return" + ? ({ kind: "assign", target: { kind: "name", name: target }, value: statement.value! } as TStmt) + : statement, + ), + }, + ]; + } + + if (init.op === "seq.map" || init.op === "seq.filter") { + const [list, fn] = init.args; + if (fn === undefined || fn.kind !== "lambda" || list === undefined) return undefined; + const element = this.spec.naming.value(fn.params[0]!.name); + const body = this.inlineLambdaBody(fn, []); + const produced = body.find((statement) => statement.kind === "return"); + if (produced === undefined || produced.kind !== "return" || produced.value === undefined) return undefined; + const inner: TStmt[] = + init.op === "seq.map" + ? [{ kind: "expr", expr: this.appendCall({ kind: "name", name: target }, produced.value) }] + : [ + { + kind: "if", + test: produced.value, + then: [ + { + kind: "expr", + expr: this.appendCall( + { kind: "name", name: target }, + { kind: "name", name: element }, + ), + }, + ], + otherwise: [], + }, + ]; + return [ + { kind: "let", name: target, type, init: { kind: "list", items: [], type }, mutable: true }, + { + kind: "forEach", + name: element, + type: fn.params[0]!.type, + iterable: this.expr(list), + body: inner, + }, + ]; + } + + return undefined; + } + + private appendCall(list: TExpr, value: TExpr): TExpr { + const selection = this.spec.table.select("seq.push", []); + return selection.candidate.emit([list, value], [], this.context()); + } + + /** Lowers a lambda body, renaming captured parameters to the caller's names. */ + private inlineLambdaBody( + lambda: Extract, + renames: readonly { from: string; to: string }[], + ): TStmt[] { + const body = this.block(lambda.body); + if (renames.length === 0) return body; + const map = new Map(renames.map((rename) => [this.spec.naming.value(rename.from), rename.to])); + const rename = (expr: TExpr): TExpr => + expr.kind === "name" && map.has(expr.name) ? { ...expr, name: map.get(expr.name)! } : expr; + return mapNames(body, rename); + } + + /* ---------------------------------------------------------------- * + * Expressions + * ---------------------------------------------------------------- */ + + /** + * Lifts a constant table out of the function that uses it. + * + * A weight table written inline would be rebuilt on every call in every target; as a module + * level constant it is literal data, which is allocated once and stays tree-shakeable. + */ + private hoistConstantTables(body: readonly TStmt[]): TStmt[] { + return mapExprs(body, (expr) => { + if (expr.kind !== "lit" || !Array.isArray(expr.value) || expr.value.length < 2) return expr; + const key = JSON.stringify(expr.value, (_key, item: unknown) => + typeof item === "bigint" ? item.toString() : item, + ); + const existing = this.constantNames.get(key); + if (existing !== undefined) return { kind: "name", name: existing }; + // The module is part of the name because Go puts every generated file in one package, + // so two modules that each hoisted a bare `table1` would redeclare it. Naming by module + // rather than by a running count also keeps an unrelated new utility from renumbering + // the tables of every file that already had one. + const scope = this.currentModule.replaceAll("/", "-"); + const name = this.spec.naming.value(`${scope}-table-${this.constants.length + 1}`); + this.constantNames.set(key, name); + this.constants.push({ name, type: expr.type, value: expr }); + return { kind: "name", name }; + }); + } + + private context(): EmitContext { + return { + require: (module) => this.imports.get(this.currentModule)?.add(module), + nameOf: (qualified) => this.names.get(qualified) ?? qualified, + needSource: (qualified) => this.needed.add(qualified), + env: () => ({ kind: "name", name: ENV_PARAM }), + }; + } + + private temp(): string { + this.temporaries += 1; + return this.spec.naming.value(`tmp${this.temporaries}`); + } + + expr(expr: CExpr): TExpr { + switch (expr.kind) { + case "lit": + return { kind: "lit", value: expr.value, type: expr.type }; + case "local": + return { + kind: "name", + name: this.spec.naming.value(expr.name), + borrowed: this.currentBorrowedParams.has(expr.name), + }; + case "none": + return { kind: "none", type: expr.type }; + case "some": + return { kind: "some", inner: this.expr(expr.inner) }; + case "record": + // A record built inline still needs its declaration in scope, which for an + // engine-defined record means importing it from the target's support file. + if (BUILTIN_RECORD_NAMES.includes(expr.typeName)) { + const perModule = this.moduleBuiltins.get(this.currentModule) ?? new Set(); + perModule.add(expr.typeName); + this.moduleBuiltins.set(this.currentModule, perModule); + } + return { + kind: "record", + typeName: this.spec.naming.type(expr.typeName), + fields: expr.fields.map((field) => ({ + name: this.spec.naming.field(field.name, true), + value: this.expr(field.value), + })), + }; + case "field": + return { + kind: "member", + target: this.expr(expr.target), + name: this.spec.naming.field(expr.name, true), + }; + case "list": + return { kind: "list", items: expr.items.map((item) => this.expr(item)), type: expr.type }; + case "call": { + const callee = this.program.functions.get(expr.fn); + const args = expr.args.map((arg) => this.expr(arg)); + const borrowedThere = this.options.borrows?.get(expr.fn); + const borrowedArgs = + borrowedThere === undefined + ? undefined + : callee?.params.map((param) => borrowedThere.has(param.name)); + if (callee?.usesEnv === true) args.push({ kind: "name", name: ENV_PARAM }); + const call: TExpr = { + kind: "call", + callee: { kind: "name", name: this.names.get(expr.fn) ?? expr.fn }, + args, + borrowedArgs, + await: this.spec.asyncColouring && callee?.effects.http === true, + }; + if (this.spec.errorsAsValues && (callee?.effects.fail.length ?? 0) > 0) { + return this.hoistFallible(call, callee!.ret); + } + return call; + } + case "op": + return this.operation(expr); + case "lambda": { + // A lambda has its own result: the enclosing function's error contract does not + // apply inside it, which matters for a target that returns errors as values. + const outerFails = this.currentFails; + const outerReturn = this.currentReturn; + this.currentFails = []; + this.currentReturn = expr.type.kind === "Lambda" ? expr.type.ret : tBool; + const body = this.block(expr.body); + this.currentFails = outerFails; + this.currentReturn = outerReturn; + return { + kind: "lambda", + params: expr.params.map((param) => ({ + name: this.spec.naming.value(param.name), + type: param.type, + })), + body, + ret: expr.type.kind === "Lambda" ? expr.type.ret : tBool, + }; + } + case "cond": { + if (!this.spec.statementTernary) { + return { + kind: "ternary", + test: this.expr(expr.test), + then: this.expr(expr.then), + otherwise: this.expr(expr.otherwise), + }; + } + // Go has no conditional expression, so the value is computed by statements. + const name = this.temp(); + const test = this.expr(expr.test); + this.pending.push({ + kind: "let", + name, + type: expr.type, + init: { kind: "zero", type: expr.type }, + mutable: true, + }); + this.pending.push({ + kind: "if", + test, + then: [{ kind: "assign", target: { kind: "name", name }, value: this.expr(expr.then) }], + otherwise: [{ kind: "assign", target: { kind: "name", name }, value: this.expr(expr.otherwise) }], + }); + return { kind: "name", name }; + } + case "and": + return { kind: "binary", op: "&&", left: this.expr(expr.left), right: this.expr(expr.right) }; + case "or": + return { kind: "binary", op: "||", left: this.expr(expr.left), right: this.expr(expr.right) }; + case "not": + return { kind: "unary", op: "!", operand: this.expr(expr.operand) }; + default: { + const exhaustive: never = expr; + return exhaustive; + } + } + } + + /** + * A chain of `+` on strings is nested `str.concat` ops, two at a time (`(a + b) + c` is + * `str.concat(str.concat(a, b), c)`) — sound, but a target whose `str.concat` allocates a fresh + * buffer per call (Rust: `engine/docs/progress.md` §8) then pays for one reallocation-and-copy + * per piece instead of one buffer sized once. Flattening the chain into its leaves before + * lowering, and handing them to `str.concatAll` when a target declares one, is how that target + * gets to make that one-buffer decision; a target with no such candidate (every one but Rust, + * today) falls straight through to the ordinary pairwise path below, unchanged. + */ + private operation(expr: Extract): TExpr { + if (expr.op === "str.concat" && this.spec.table.has("str.concatAll")) { + const pieces = flattenConcat(expr); + if (pieces.length > 2) { + return this.emitOp("str.concatAll", pieces.map((piece) => this.expr(piece)), pieces.map((piece) => piece.type), undefined); + } + } + const op = expr.op === "re.test" ? "re.test" : expr.op; + return this.emitOp(op, expr.args.map((arg) => this.expr(arg)), expr.args.map((arg) => arg.type), expr.regex); + } + + private emitOp(op: string, args: readonly TExpr[], types: readonly SemType[], regex: NormalizedRegex | undefined): TExpr { + const selection = this.spec.table.select(op, types); + const candidate = selection.candidate; + if (candidate.sourceFn !== undefined) { + this.needed.add(candidate.sourceFn); + const set = this.moduleNeeds.get(this.currentModule); + if (set === undefined) this.moduleNeeds.set(this.currentModule, new Set([candidate.sourceFn])); + else set.add(candidate.sourceFn); + } + const context = this.context(); + if (regex !== undefined) return candidate.emit(args, types, { ...context, regex }); + return candidate.emit(args, types, context); + } + + /** `v, err := f(…)` plus the early return every Go caller writes by hand. */ + private hoistFallible(call: TExpr, type: SemType): TExpr { + const value = this.temp(); + const error = `${value}Err`; + this.pending.push({ + kind: "multiLet", + names: [value, error], + types: [type, tString("ascii")], + init: call, + }); + this.pending.push({ + kind: "if", + test: { kind: "binary", op: "!=", left: { kind: "name", name: error }, right: { kind: "raw", text: "nil" } }, + then: [ + { + kind: "return", + value: { kind: "zero", type: this.currentReturn }, + extra: [{ kind: "name", name: error }], + }, + ], + otherwise: [], + }); + return { kind: "name", name: value }; + } +} + +/** Renames free identifiers in a statement list. */ +function mapNames(body: readonly TStmt[], rename: (expr: TExpr) => TExpr): TStmt[] { + return mapExprs(body, rename); +} + +/** The leaves of a `str.concat` chain, left to right — see `operation`'s comment on why. */ +function flattenConcat(expr: CExpr): CExpr[] { + if (expr.kind === "op" && expr.op === "str.concat") { + return [...flattenConcat(expr.args[0]!), ...flattenConcat(expr.args[1]!)]; + } + return [expr]; +} + +export type { TExpr, TStmt, TFunc, TModule }; diff --git a/engine/src/backend/select.ts b/engine/src/backend/select.ts new file mode 100644 index 000000000..2214349c5 --- /dev/null +++ b/engine/src/backend/select.ts @@ -0,0 +1,190 @@ +/** + * Lowering selection. + * + * Each target declares, per intrinsic, the ways it could implement that operation, what it needs + * to be allowed to (facts on the argument types, a language baseline, a dependency) and what it + * costs. Selection is deterministic and explainable, and the compiler writes the explanation to + * `LOWERING.md` so a reviewer sees why a native lowering was or was not taken. + */ + +import type { NormalizedRegex } from "../regex.ts"; +import type { SemType } from "../types.ts"; +import { typeToString } from "../types.ts"; +import type { TExpr } from "./tast.ts"; + +export type CostClass = { + readonly alloc: "none" | "one" | "many"; + readonly time: "constant" | "linear" | "nlogn" | "quadratic"; +}; + +export type Impl = "native" | "library" | "portable"; + +export type EmitContext = { + /** Records a module the emitted code needs (an import line in the generated file). */ + readonly require: (module: string) => void; + /** Target identifier of a generated core function, used by portable lowerings. */ + readonly nameOf: (qualified: string) => string; + /** Marks a source library function as needed, so it lands in the dependency closure. */ + readonly needSource: (qualified: string) => void; + /** The expression holding the capability record inside the current function. */ + readonly env: () => TExpr; + /** The normalized pattern, for the `re.test` lowerings. */ + readonly regex?: NormalizedRegex; +}; + +export type Candidate = { + readonly op: string; + readonly impl: Impl; + /** The preconditions the argument types must satisfy for this lowering to be admissible. */ + readonly requires?: (args: readonly SemType[]) => boolean; + /** Why the precondition exists, printed in `LOWERING.md`. */ + readonly because?: string; + readonly deps?: readonly string[]; + readonly cost: CostClass; + readonly emit: (args: readonly TExpr[], types: readonly SemType[], ctx: EmitContext) => TExpr; + /** For a portable lowering: the source library function that implements the operation. */ + readonly sourceFn?: string; +}; + +const ALLOC_RANK: Record = { none: 0, one: 1, many: 2 }; +const TIME_RANK: Record = { + constant: 0, + linear: 1, + nlogn: 2, + quadratic: 3, +}; +const IMPL_RANK: Record = { native: 0, library: 1, portable: 2 }; + +export type Selection = { + readonly candidate: Candidate; + readonly reason: string; + readonly rejected: readonly { readonly impl: Impl; readonly because: string }[]; +}; + +export class LoweringTable { + private readonly byOp = new Map(); + readonly selections: { op: string; args: string; impl: Impl; reason: string }[] = []; + + constructor(candidates: readonly Candidate[]) { + for (const candidate of candidates) { + const existing = this.byOp.get(candidate.op); + if (existing === undefined) this.byOp.set(candidate.op, [candidate]); + else existing.push(candidate); + } + } + + has(op: string): boolean { + return this.byOp.has(op); + } + + /** + * Applies the rules lexicographically: admissible candidates first, then the lower declared + * cost class, then native before library before portable, then declaration order. + */ + select(op: string, args: readonly SemType[]): Selection { + const candidates = this.byOp.get(op) ?? []; + const rejected: { impl: Impl; because: string }[] = []; + const admissible = candidates.filter((candidate, index) => { + const ok = candidate.requires === undefined || candidate.requires(args); + if (!ok) { + rejected.push({ + impl: candidate.impl, + because: candidate.because ?? `candidate ${index} precondition not met`, + }); + } + return ok; + }); + if (admissible.length === 0) { + throw new Error( + `no admissible lowering for ${op}(${args.map(typeToString).join(", ")})${ + rejected.length === 0 ? "" : `; rejected: ${rejected.map((item) => item.because).join("; ")}` + }`, + ); + } + const ordered = [...admissible].sort((left, right) => { + const alloc = ALLOC_RANK[left.cost.alloc] - ALLOC_RANK[right.cost.alloc]; + if (alloc !== 0) return alloc; + const time = TIME_RANK[left.cost.time] - TIME_RANK[right.cost.time]; + if (time !== 0) return time; + return IMPL_RANK[left.impl] - IMPL_RANK[right.impl]; + }); + const chosen = ordered[0]!; + // An unopposed candidate still has a reason, and it used to be dropped: `LOWERING.md` is + // supposed to record the rule that decided every selection, not only the ones that had a + // rival to reject. + const why = chosen.because === undefined ? "" : `; ${chosen.because}`; + const reason = + rejected.length === 0 + ? `only candidate, cost ${chosen.cost.alloc}/${chosen.cost.time}${why}` + : `${chosen.impl}, cost ${chosen.cost.alloc}/${chosen.cost.time}${why}; rejected ${rejected.map((item) => item.because).join(", ")}`; + this.selections.push({ op, args: args.map(typeToString).join(", "), impl: chosen.impl, reason }); + return { candidate: chosen, reason, rejected }; + } + + /** The reviewable record of every non-trivial selection. */ + renderLoweringDoc(target: string): string { + const seen = new Map(); + for (const selection of this.selections) { + seen.set(`${selection.op}(${selection.args})`, selection); + } + const rows = [...seen.values()].sort((left, right) => + `${left.op}(${left.args})`.localeCompare(`${right.op}(${right.args})`), + ); + const lines = [ + `# Lowering selections — ${target}`, + "", + "Generated by the engine. Each row is one operation, the argument types it was called", + "with, the implementation that was selected, and the rule that decided it.", + "", + "| operation | argument types | implementation | why |", + "| --- | --- | --- | --- |", + ]; + for (const selection of rows) { + lines.push( + `| \`${selection.op}\` | \`${cell(selection.args)}\` | ${selection.impl} | ${cell(selection.reason)} |`, + ); + } + lines.push(""); + const counts = { native: 0, library: 0, portable: 0 }; + for (const selection of seen.values()) counts[selection.impl]++; + lines.push( + `Mix: ${counts.native} native, ${counts.library} library, ${counts.portable} portable.`, + "", + ); + return lines.join("\n"); + } +} + +/** A table cell: an enum type contains `|`, which would otherwise end the column. */ +function cell(text: string): string { + return text.replaceAll("|", "\\|"); +} + +/* ------------------------------------------------------------------ * + * Fact helpers used by the capability tables + * ------------------------------------------------------------------ */ + +export function argIsAscii(index: number) { + return (args: readonly SemType[]): boolean => { + const arg = args[index]; + return arg !== undefined && arg.kind === "String" && arg.cls !== "none"; + }; +} + +export function argIsDigits(index: number) { + return (args: readonly SemType[]): boolean => { + const arg = args[index]; + return arg !== undefined && arg.kind === "String" && arg.cls === "digits"; + }; +} + +export function argFitsSafeInt(index: number) { + return (args: readonly SemType[]): boolean => { + const arg = args[index]; + return arg !== undefined && arg.kind === "Int" && arg.lo >= -(2n ** 53n - 1n) && arg.hi <= 2n ** 53n - 1n; + }; +} + +export function always(): boolean { + return true; +} diff --git a/engine/src/backend/tast.ts b/engine/src/backend/tast.ts new file mode 100644 index 000000000..c530bf43c --- /dev/null +++ b/engine/src/backend/tast.ts @@ -0,0 +1,322 @@ +/** + * The Target AST. + * + * Generated code is built as a tree and printed once; nothing in the engine concatenates source + * text. The tree is shared by the three backends because the constructs they need overlap almost + * entirely; what differs — statement-only languages, multi-return errors, async colouring — is + * expressed as nanopasses over this tree, and the per-target printer decides the syntax. + */ + +import type { SemType } from "../types.ts"; +import type { Value } from "../values.ts"; + +export type TExpr = + | { readonly kind: "lit"; readonly value: Value; readonly type: SemType } + | { + readonly kind: "name"; + readonly name: string; + /** + * Set by a target's borrow-aware lowering (Rust's — see `docs/decisions/0010-*.md`) when + * this name is a reference (`&str`/`&[T]`), not an owned value, so the printer never has + * to guess from ambient state whether wrapping it in another `&` would double-borrow. A + * target that never borrows parameters (Go, Python, TypeScript) leaves this unset. + */ + readonly borrowed?: boolean; + } + | { + readonly kind: "call"; + readonly callee: TExpr; + readonly args: readonly TExpr[]; + readonly await?: boolean; + /** + * Parallel to `args`: whether the callee's parameter at that position was found borrowable + * (see `docs/decisions/0010-*.md`), so a borrow-aware printer can pass a reference instead + * of cloning. Set once, during lowering, from the whole-program borrow map — never decided + * by the printer itself. Unset (or `undefined` per position) means "owned", the only + * meaning every non-Rust target has ever needed. + */ + readonly borrowedArgs?: readonly boolean[]; + } + | { + readonly kind: "method"; + readonly target: TExpr; + readonly name: string; + readonly args: readonly TExpr[]; + readonly await?: boolean; + } + | { readonly kind: "member"; readonly target: TExpr; readonly name: string } + | { readonly kind: "index"; readonly target: TExpr; readonly index: TExpr } + | { readonly kind: "binary"; readonly op: string; readonly left: TExpr; readonly right: TExpr } + | { readonly kind: "unary"; readonly op: string; readonly operand: TExpr } + | { readonly kind: "ternary"; readonly test: TExpr; readonly then: TExpr; readonly otherwise: TExpr } + | { readonly kind: "list"; readonly items: readonly TExpr[]; readonly type: SemType } + | { + readonly kind: "record"; + readonly typeName: string; + readonly fields: readonly { readonly name: string; readonly value: TExpr }[]; + } + | { + readonly kind: "lambda"; + readonly params: readonly { readonly name: string; readonly type: SemType }[]; + readonly body: readonly TStmt[]; + readonly ret: SemType; + } + | { readonly kind: "none"; readonly type: SemType } + | { readonly kind: "some"; readonly inner: TExpr } + /** A target-specific fragment produced by that target's capability table. */ + | { readonly kind: "raw"; readonly text: string; readonly precedence?: number } + /** The target's zero value for a type: what a Go function returns beside a non-nil error. */ + | { readonly kind: "zero"; readonly type: SemType }; + +export type TStmt = + | { + readonly kind: "let"; + readonly name: string; + readonly type: SemType; + readonly init: TExpr; + readonly mutable: boolean; + } + | { readonly kind: "assign"; readonly target: TExpr; readonly value: TExpr } + | { readonly kind: "if"; readonly test: TExpr; readonly then: readonly TStmt[]; readonly otherwise: readonly TStmt[] } + | { + readonly kind: "switch"; + readonly subject: TExpr; + readonly cases: readonly { readonly values: readonly Value[]; readonly body: readonly TStmt[] }[]; + readonly otherwise?: readonly TStmt[]; + } + | { + readonly kind: "for"; + readonly name: string; + readonly type: SemType; + readonly from: TExpr; + readonly to: TExpr; + readonly inclusive: boolean; + readonly step: bigint; + readonly body: readonly TStmt[]; + } + | { + readonly kind: "forEach"; + readonly name: string; + readonly type: SemType; + readonly iterable: TExpr; + readonly body: readonly TStmt[]; + } + | { + readonly kind: "multiLet"; + readonly names: readonly string[]; + readonly types: readonly SemType[]; + readonly init: TExpr; + } + | { readonly kind: "return"; readonly value?: TExpr; readonly extra?: readonly TExpr[] } + | { readonly kind: "throw"; readonly errorClass: string; readonly args: readonly TExpr[] } + | { readonly kind: "break" } + | { readonly kind: "continue" } + | { readonly kind: "expr"; readonly expr: TExpr } + | { readonly kind: "raw"; readonly text: string }; + +export type TParam = { + readonly name: string; + readonly type: SemType; + readonly doc?: string; + /** Set by a borrow-aware lowering when this parameter may be declared `&str`/`&[T]`. */ + readonly borrowed?: boolean; +}; + +export type TFunc = { + readonly name: string; + readonly params: readonly TParam[]; + readonly ret: SemType; + readonly body: readonly TStmt[]; + /** A utility: the published core API. See `CFunc.exported` — unrelated to `moduleExported`. */ + readonly exported: boolean; + /** The source module's own `export` keyword; what a printer turns into its own notion of + * public (`export`, `pub`, no leading underscore). See `CFunc.moduleExported`. */ + readonly moduleExported: boolean; + readonly doc?: string; + readonly isAsync: boolean; + /** Domain errors this function may raise; the Go backend turns these into a second return. */ + readonly fails: readonly string[]; + readonly usesEnv: boolean; + /** + * True for the capability-taking form of an entry point that also has a no-capabilities public + * wrapper, or for an entry point that takes capabilities directly because its target has no + * default to wrap it with. `generate()` uses this to point the differential driver at the form + * that takes fakes and to mark it in `API.json`, rather than listing it as an ordinary utility. + */ + readonly seam?: boolean; + /** + * Set on a synthesized public wrapper (see `docs/decisions/0011-*.md`): the name of the seam + * function in the same module it delegates to. `generate()` uses this to keep the wrapper out + * of the driver's dispatch table — it cannot accept injected fakes, having no capability + * parameter at all. + */ + readonly seamName?: string; + /** Set on a seam (the mirror of `seamName`): the public wrapper's own name, for `API.json`. */ + readonly wrapperName?: string; + /** Provenance: the source module and function this was generated from. */ + readonly source: { readonly module: string; readonly name: string; readonly start: number; readonly end: number }; +}; + +export type TRecord = { + readonly name: string; + readonly fields: readonly { readonly name: string; readonly type: SemType; readonly doc?: string }[]; + readonly doc?: string; +}; + +export type TErrorClass = { + readonly name: string; + readonly base?: string; + readonly doc?: string; +}; + +export type TConst = { + readonly name: string; + readonly type: SemType; + readonly value: TExpr; +}; + +export type TModule = { + /** Path of the generated file, relative to the target's output directory. */ + readonly path: string; + /** The Core module this file was generated from, which the import computation needs. */ + readonly sourcePath: string; + readonly imports: readonly TImport[]; + readonly records: readonly TRecord[]; + readonly errors: readonly TErrorClass[]; + /** Constant tables hoisted out of the functions that use them. */ + readonly constants: readonly TConst[]; + readonly functions: readonly TFunc[]; + readonly header: string; + /** Target modules the emitted code needs, collected from the capability table. */ + readonly requires: readonly string[]; +}; + +export type TImport = { + readonly from: string; + readonly names: readonly string[]; + /** True when the names are types only, which a type-stripping runtime has to be told. */ + readonly typeOnly?: boolean; + /** A whole-module import, used by Python and Go. */ + readonly module?: boolean; +}; + + +/** Escapes every non-ASCII scalar, so a generated file is plain ASCII and never carries a BOM. */ +export function asciiString(value: string): string { + let out = '"'; + for (const scalar of value) { + const point = scalar.codePointAt(0)!; + if (scalar === '"') out += '\\"'; + else if (scalar === "\\") out += "\\\\"; + else if (point === 0x0a) out += "\\n"; + else if (point === 0x0d) out += "\\r"; + else if (point === 0x09) out += "\\t"; + else if (point < 0x20 || point > 0x7e) { + out += point > 0xffff ? `\\U${point.toString(16).padStart(8, "0")}` : `\\u${point.toString(16).padStart(4, "0")}`; + } else out += scalar; + } + return `${out}"`; +} + +/** Rewrites every expression of a statement list bottom-up. */ +export function mapExprs(body: readonly TStmt[], visit: (expr: TExpr) => TExpr): TStmt[] { + const expr = (node: TExpr): TExpr => { + const mapped = ((): TExpr => { + switch (node.kind) { + case "call": + return { ...node, callee: expr(node.callee), args: node.args.map(expr) }; + case "method": + return { ...node, target: expr(node.target), args: node.args.map(expr) }; + case "member": + return { ...node, target: expr(node.target) }; + case "index": + return { ...node, target: expr(node.target), index: expr(node.index) }; + case "binary": + return { ...node, left: expr(node.left), right: expr(node.right) }; + case "unary": + return { ...node, operand: expr(node.operand) }; + case "ternary": + return { ...node, test: expr(node.test), then: expr(node.then), otherwise: expr(node.otherwise) }; + case "list": + return { ...node, items: node.items.map(expr) }; + case "record": + return { ...node, fields: node.fields.map((field) => ({ ...field, value: expr(field.value) })) }; + case "lambda": + return { ...node, body: mapExprs(node.body, visit) }; + case "some": + return { ...node, inner: expr(node.inner) }; + default: + return node; + } + })(); + return visit(mapped); + }; + + return body.map((statement): TStmt => { + switch (statement.kind) { + case "let": + return { ...statement, init: expr(statement.init) }; + case "assign": + return { ...statement, target: expr(statement.target), value: expr(statement.value) }; + case "if": + return { + ...statement, + test: expr(statement.test), + then: mapExprs(statement.then, visit), + otherwise: mapExprs(statement.otherwise, visit), + }; + case "switch": + return { + ...statement, + subject: expr(statement.subject), + cases: statement.cases.map((entry) => ({ ...entry, body: mapExprs(entry.body, visit) })), + otherwise: statement.otherwise === undefined ? undefined : mapExprs(statement.otherwise, visit), + }; + case "for": + return { + ...statement, + from: expr(statement.from), + to: expr(statement.to), + body: mapExprs(statement.body, visit), + }; + case "forEach": + return { ...statement, iterable: expr(statement.iterable), body: mapExprs(statement.body, visit) }; + case "multiLet": + return { ...statement, init: expr(statement.init) }; + case "return": + return { + ...statement, + value: statement.value === undefined ? undefined : expr(statement.value), + extra: statement.extra?.map(expr), + }; + case "throw": + return { ...statement, args: statement.args.map(expr) }; + case "expr": + return { ...statement, expr: expr(statement.expr) }; + default: + return statement; + } + }); +} + +/** Rewrites every statement list of a statement list top-down. */ +export function mapStmts(body: readonly TStmt[], visit: (statements: readonly TStmt[]) => TStmt[]): TStmt[] { + const inner = body.map((statement): TStmt => { + switch (statement.kind) { + case "if": + return { ...statement, then: mapStmts(statement.then, visit), otherwise: mapStmts(statement.otherwise, visit) }; + case "switch": + return { + ...statement, + cases: statement.cases.map((entry) => ({ ...entry, body: mapStmts(entry.body, visit) })), + otherwise: statement.otherwise === undefined ? undefined : mapStmts(statement.otherwise, visit), + }; + case "for": + case "forEach": + return { ...statement, body: mapStmts(statement.body, visit) }; + default: + return statement; + } + }); + return visit(inner); +} diff --git a/engine/src/cli.ts b/engine/src/cli.ts new file mode 100644 index 000000000..34c2bdbe0 --- /dev/null +++ b/engine/src/cli.ts @@ -0,0 +1,111 @@ +#!/usr/bin/env node +/** + * The engine's command line. + * + * logic-engine check [--project ] + * logic-engine dump [--project ] [--stage hir|core] + * logic-engine build [--project ] [--target ] [--no-idioms] [--out ] + */ + +import { readFileSync } from "node:fs"; +import { resolve } from "node:path"; +import { CompileError, compileProject, dumpProgram, renderDiagnostic } from "./api.ts"; +import { dumpHir } from "./hir/ast.ts"; +import { generate, writeFiles } from "./backend/generate.ts"; +import { formatOutput } from "./backend/format.ts"; +import { loadConfig } from "./project.ts"; +import type { TargetName } from "./project.ts"; +import { TYPESCRIPT_BACKEND } from "./targets/typescript/index.ts"; +import { PYTHON_BACKEND } from "./targets/python/index.ts"; +import { GO_BACKEND } from "./targets/go/index.ts"; +import { RUST_BACKEND } from "./targets/rust/index.ts"; + +const BACKENDS = { + typescript: TYPESCRIPT_BACKEND, + python: PYTHON_BACKEND, + go: GO_BACKEND, + rust: RUST_BACKEND, +}; + +function flag(name: string, fallback?: string): string | undefined { + const index = process.argv.indexOf(`--${name}`); + return index === -1 ? fallback : process.argv[index + 1]; +} + +function has(name: string): boolean { + return process.argv.includes(`--${name}`); +} + +function main(): void { + const command = process.argv[2] ?? "build"; + const projectDir = resolve(flag("project", ".")!); + const { config, root } = loadConfig(resolve(projectDir, "engine.config.json")); + const sourceRoot = resolve(root, config.sourceRoot); + + try { + const compilation = compileProject(sourceRoot, { noOptimize: has("no-optimize") }); + + if (command === "check") { + process.stdout.write(`ok: ${compilation.program.functions.size} functions, ${compilation.program.entryPoints.length} utilities\n`); + return; + } + + if (command === "dump") { + if (flag("stage", "core") === "hir") { + for (const module of compilation.modules) process.stdout.write(`${dumpHir(module)}\n`); + } else { + process.stdout.write(`${dumpProgram(compilation.program)}\n`); + } + return; + } + + if (command !== "build") { + process.stderr.write(`unknown command ${command}\n`); + process.exitCode = 1; + return; + } + + const targets = (flag("target") === undefined ? config.targets : [flag("target") as TargetName]).filter( + (name): name is TargetName => name in BACKENDS, + ); + const outRoot = resolve(root, flag("out", config.out)!); + const noIdioms = has("no-idioms"); + + for (const target of targets) { + const backend = BACKENDS[target]; + const result = generate(compilation.program, backend, { noIdioms }); + const outDir = resolve(outRoot, noIdioms ? `${target}-plain` : target); + writeFiles(outDir, result, { + "LOWERING.md": result.lowering, + "API.json": `${JSON.stringify(result.api, null, "\t")}\n`, + "SOURCEMAP.json": `${JSON.stringify(result.sourceMap, null, "\t")}\n`, + }); + const formatted = has("no-format") ? { applied: [], missing: [] } : formatOutput(target, outDir); + process.stdout.write( + `${target}: ${result.files.length} files -> ${outDir}${formatted.applied.length === 0 ? "" : ` (${formatted.applied.join(", ")})`}\n`, + ); + // Unformatted output is still correct, but it is not the bytes the committed output + // holds, so say which tool is missing rather than leave a diff to explain it. + if (formatted.missing.length > 0) { + process.stderr.write(`warning: ${target} left unformatted, not installed: ${formatted.missing.join(", ")}\n`); + } + } + } catch (error) { + if (error instanceof CompileError) { + for (const diagnostic of error.diagnostics) { + let source: string | undefined; + try { + source = readFileSync(diagnostic.span.file, "utf8"); + } catch { + source = undefined; + } + process.stderr.write(`${renderDiagnostic(diagnostic, source)}\n\n`); + } + process.exitCode = 1; + return; + } + throw error; + } +} + +main(); diff --git a/engine/src/comptime/eval.ts b/engine/src/comptime/eval.ts new file mode 100644 index 000000000..1d9016c2a --- /dev/null +++ b/engine/src/comptime/eval.ts @@ -0,0 +1,111 @@ +/** + * Comptime evaluation. + * + * Runs whenever every input of a computation is known at compile time: constant tables, baked + * datasets, normalized regexes, folded arithmetic. It is deterministic and side-effect free by + * construction — no capability is reachable from here — and its output is data embedded in the + * Core. + */ + +import { lookupIntrinsic } from "../intrinsics/index.ts"; +import type { CExpr } from "../core/ir.ts"; +import { NONE, record, some, valuesEqual } from "../values.ts"; +import type { Value } from "../values.ts"; +import { regexMatches } from "../regex.ts"; +import { codePointsOf } from "../values.ts"; + +export class NotConstant extends Error {} + +const FORBIDDEN_CONTEXT = { + http: () => { + throw new NotConstant("Http is not reachable at compile time"); + }, + now: () => { + throw new NotConstant("Clock is not reachable at compile time"); + }, + sleep: () => { + throw new NotConstant("Clock is not reachable at compile time"); + }, + nextU32: () => { + throw new NotConstant("Random is not reachable at compile time"); + }, +}; + +/** Evaluates an expression whose value is fully known, or throws `NotConstant`. */ +export function evalConst(expr: CExpr): Value { + switch (expr.kind) { + case "lit": + return expr.value; + case "none": + return NONE; + case "some": + return some(evalConst(expr.inner)); + case "list": + return expr.items.map((item) => evalConst(item)); + case "record": { + const fields: Record = {}; + for (const field of expr.fields) fields[field.name] = evalConst(field.value); + return record(expr.typeName, fields); + } + case "field": { + const target = evalConst(expr.target); + if (typeof target === "object" && target !== null && "__kind" in target && target.__kind === "record") { + return target.fields[expr.name]!; + } + throw new NotConstant("field access on a non-record"); + } + case "op": { + if (expr.op === "re.retain") { + const subject = String(evalConst(expr.args[0]!)); + return codePointsOf(subject) + .filter((point) => regexMatches(expr.regex!, [point])) + .map((point) => String.fromCodePoint(point)) + .join(""); + } + if (expr.op === "re.test") { + const subject = evalConst(expr.args[0]!); + return regexMatches(expr.regex!, codePointsOf(String(subject))); + } + const intrinsic = lookupIntrinsic(expr.op); + if (intrinsic === undefined || !intrinsic.comptime) { + throw new NotConstant(`${expr.op} is not available at compile time`); + } + return intrinsic.evaluate( + expr.args.map((arg) => evalConst(arg)), + FORBIDDEN_CONTEXT, + ); + } + case "not": + return !(evalConst(expr.operand) === true); + case "and": { + const left = evalConst(expr.left); + return left === true ? evalConst(expr.right) : false; + } + case "or": { + const left = evalConst(expr.left); + return left === true ? true : evalConst(expr.right); + } + case "cond": + return evalConst(expr.test) === true ? evalConst(expr.then) : evalConst(expr.otherwise); + default: + throw new NotConstant(`${expr.kind} is not a constant expression`); + } +} + +/** Whether two constant expressions denote the same value; used by the optimizer. */ +export function sameConstant(left: CExpr, right: CExpr): boolean { + try { + return valuesEqual(evalConst(left), evalConst(right)); + } catch { + return false; + } +} + +export function tryEvalConst(expr: CExpr): Value | undefined { + try { + return evalConst(expr); + } catch (error) { + if (error instanceof NotConstant) return undefined; + throw error; + } +} diff --git a/engine/src/conformance/differential.ts b/engine/src/conformance/differential.ts new file mode 100644 index 000000000..3de00287a --- /dev/null +++ b/engine/src/conformance/differential.ts @@ -0,0 +1,186 @@ +/** + * Differential conformance. + * + * The reference interpreter and every generated target answer the same cases through the same + * JSON protocol, and the harness compares them. Targets run in batches — one process per + * language, JSON lines over stdin and stdout — so the cost of a case is one line, not one process. + */ + +import { spawnSync } from "node:child_process"; +import type { CProgram } from "../core/ir.ts"; +import { Interpreter } from "../interp/interp.ts"; +import type { Capabilities } from "../interp/interp.ts"; +import { DomainFailure } from "../intrinsics/index.ts"; +import type { SemType } from "../types.ts"; +import type { Value } from "../values.ts"; +import { NONE, civilDate, decimal, record, some } from "../values.ts"; + +export type Case = { + readonly fn: string; + readonly args: readonly Value[]; + /** A human readable label for the report. */ + readonly label?: string; +}; + +export type Outcome = { readonly ok: true; readonly value: unknown } | { readonly ok: false; readonly error: string }; + +export type TargetRunner = { + readonly name: string; + readonly command: string; + readonly args: readonly string[]; + readonly cwd: string; +}; + +/** Converts a semantic value into the plain JSON every generated driver speaks. */ +export function toJson(value: Value): unknown { + if (typeof value === "bigint") return Number(value); + if (Array.isArray(value)) return value.map(toJson); + if (typeof value === "object" && value !== null) { + const tagged = value as { __kind: string }; + if (tagged.__kind === "none") return null; + if (tagged.__kind === "some") return toJson((tagged as unknown as { value: Value }).value); + if (tagged.__kind === "record") { + const fields = (tagged as unknown as { fields: Record }).fields; + const out: Record = {}; + for (const [key, item] of Object.entries(fields)) out[key] = toJson(item); + return out; + } + if (tagged.__kind === "decimal") return Number((tagged as unknown as { unscaled: bigint }).unscaled); + if (tagged.__kind === "date") return (tagged as unknown as { days: number }).days; + } + return value; +} + +/** Reads a driver's answer back into a semantic value, guided by the declared type. */ +export function fromJson(raw: unknown, type: SemType): Value { + switch (type.kind) { + case "Option": + return raw === null || raw === undefined ? NONE : some(fromJson(raw, type.inner)); + case "Int": + case "Decimal": + case "CivilDate": + case "Instant": + case "Duration": + return BigInt(Math.trunc(Number(raw))); + case "List": + return (raw as unknown[]).map((item) => fromJson(item, type.elem)); + case "Record": { + const fields: Record = {}; + for (const [key, item] of Object.entries(raw as Record)) { + fields[key] = item as Value; + } + return record(type.name, fields); + } + default: + return raw as Value; + } +} + +/** + * Adapts a case's arguments to the interpreter's representation. + * + * A case is written in the protocol's plain values (a civil date is a day count), while the + * interpreter keeps the semantic ones; generated targets use the plain form directly, which is + * exactly the representation freedom each backend is allowed. + */ +export function coerceValue(value: Value, type: SemType): Value { + switch (type.kind) { + case "CivilDate": + return typeof value === "bigint" ? civilDate(Number(value)) : value; + case "Decimal": + return typeof value === "bigint" ? decimal(value, type.scale) : value; + case "Option": + return value === null || (typeof value === "object" && "__kind" in value && value.__kind === "none") + ? NONE + : coerceValue(value, type.inner); + case "List": + return Array.isArray(value) ? value.map((item) => coerceValue(item, type.elem)) : value; + default: + return value; + } +} + +export function runInterpreter( + program: CProgram, + cases: readonly Case[], + capabilities?: Capabilities, +): Outcome[] { + return cases.map((testCase) => { + const interpreter = new Interpreter(program, capabilities); + const fn = program.functions.get(testCase.fn); + const args = testCase.args.map((argument, index) => { + const type = fn?.params[index]?.type; + return type === undefined ? argument : coerceValue(argument, type); + }); + try { + const value = interpreter.call(testCase.fn, args); + return { ok: true, value: toJson(value) }; + } catch (error) { + if (error instanceof DomainFailure) return { ok: false, error: error.errorType }; + throw error; + } + }); +} + +/** Runs every case through one target process. */ +export function runTarget(runner: TargetRunner, cases: readonly Case[]): Outcome[] { + const input = `${cases + .map((testCase) => JSON.stringify({ fn: testCase.fn, args: testCase.args.map(toJson) })) + .join("\n")}\n`; + const result = spawnSync(runner.command, [...runner.args], { + cwd: runner.cwd, + input, + encoding: "utf8", + maxBuffer: 256 * 1024 * 1024, + }); + if (result.status !== 0) { + throw new Error(`${runner.name} driver failed: ${result.stderr || result.stdout}`); + } + const lines = result.stdout.split("\n").filter((line) => line.trim() !== ""); + if (lines.length !== cases.length) { + throw new Error(`${runner.name} driver answered ${lines.length} of ${cases.length} cases`); + } + return lines.map((line) => JSON.parse(line) as Outcome); +} + +export type Divergence = { + readonly target: string; + readonly case: Case; + readonly expected: Outcome; + readonly actual: Outcome; +}; + +export function compare( + reference: readonly Outcome[], + actual: readonly Outcome[], + cases: readonly Case[], + target: string, +): Divergence[] { + const divergences: Divergence[] = []; + for (const [index, expected] of reference.entries()) { + const observed = actual[index]!; + if (!sameOutcome(expected, observed)) { + divergences.push({ target, case: cases[index]!, expected, actual: observed }); + } + } + return divergences; +} + +function sameOutcome(left: Outcome, right: Outcome): boolean { + if (left.ok !== right.ok) return false; + if (!left.ok || !right.ok) return (left as { error: string }).error === (right as { error: string }).error; + return JSON.stringify(normalize(left.value)) === JSON.stringify(normalize(right.value)); +} + +/** Field names differ by convention across targets, so comparison ignores case and underscores. */ +function normalize(value: unknown): unknown { + if (value === undefined) return null; + if (Array.isArray(value)) return value.map(normalize); + if (typeof value === "object" && value !== null) { + const entries = Object.entries(value as Record) + .map(([key, item]) => [key.toLowerCase().replaceAll("_", ""), normalize(item)] as const) + .sort(([left], [right]) => left.localeCompare(right)); + return Object.fromEntries(entries); + } + return value; +} diff --git a/engine/src/core/check.ts b/engine/src/core/check.ts new file mode 100644 index 000000000..82065d8cb --- /dev/null +++ b/engine/src/core/check.ts @@ -0,0 +1,3187 @@ +/** + * The semantic checker: typed HIR in, annotated Core out. + * + * Typing and lowering are one pass on purpose (docs/decisions/0003-checker-lowers.md): every + * typing decision — which intrinsic an operator resolves to, whether an index is provably in + * range, whether a string is proven ASCII — is exactly the decision the lowering has to make. + * + * Refinements are types here, not a side table, so flow-sensitive narrowing (`if (!re.test(P, s)) + * return …`) and interval analysis over loops both fall out of ordinary type checking. + */ + +import type { Diagnostics, Span } from "../diagnostics.ts"; +import { NO_SPAN } from "../diagnostics.ts"; +import type { EffectSet } from "../effects.ts"; +import { PURE, unionEffects, withFail } from "../effects.ts"; +import type { HExpr, HFunc, HModule, HStmt, HTypeExpr } from "../hir/ast.ts"; +import { BUILTIN_ERRORS, BUILTIN_RECORDS, SignatureError, lookupIntrinsic } from "../intrinsics/index.ts"; +import { normalizeRegex, RegexError } from "../regex.ts"; +import type { NormalizedRegex } from "../regex.ts"; +import type { SemType } from "../types.ts"; +import { + MAX_COLLECTION_LENGTH, + SAFE_INT_HI, + SAFE_INT_LO, + isSubtype, + join, + tBool, + tCivilDate, + tDecimal, + tDuration, + tEnum, + tFloat, + tInstant, + tInt, + tIntDefault, + tLambda, + tList, + tNever, + tOption, + tRecord, + tString, + tVoid, + typeToString, +} from "../types.ts"; +import type { Value } from "../values.ts"; +import { evalConst, tryEvalConst } from "../comptime/eval.ts"; +import { STDLIB_SURFACE } from "../stdlib.ts"; +import type { CConst, CErrorDef, CExpr, CFunc, CProgram, CRecordDef, CStmt } from "./ir.ts"; + +const INTRINSIC_MODULES = new Set([ + "str", + "seq", + "re", + "int", + "float", + "dec", + "date", + "opt", + "http", + "clock", + "random", + "task", +]); + +/** String methods the subset maps to intrinsics, and the ones it refuses with a suggestion. */ +const STRING_METHODS: Record = { + charCodeAt: "str.codeAt", + charAt: "str.charAt", + slice: "str.slice", + indexOf: "str.indexOf", + includes: "str.contains", + startsWith: "str.startsWith", + endsWith: "str.endsWith", + repeat: "str.repeat", + padStart: "str.padStart", + trim: "str.trim", + split: "str.split", + toUpperCase: "str.asciiUpper", + toLowerCase: "str.asciiLower", +}; + +/** + * `String#charCodeAt`, `#charAt` and `#slice` read the string by *position*, which only means the + * same thing in every target when the string is proven ASCII (docs/semantics.md, "Strings"); see + * `requireAsciiPositional`. + */ +const POSITIONAL_STRING_METHODS = new Set(["charCodeAt", "charAt", "slice"]); + +/** + * `String#toUpperCase`/`#toLowerCase` run the host's full Unicode case-folding table, which + * `str.asciiUpper`/`str.asciiLower`'s own doc names directly: "a proven-ASCII argument unlocks + * the host's own case mapping." Outside ASCII the two diverge — Unicode default case folding + * touches scalars an ASCII-only table leaves alone, and differs again by target — so, like the + * positional methods above, these are only the ordinary spelling of the intrinsic once the + * string is proven ASCII; see `requireAsciiCase`. + */ +const CASE_STRING_METHODS = new Set(["toUpperCase", "toLowerCase"]); + +const LIST_METHODS: Record = { + map: "seq.map", + filter: "seq.filter", + includes: "seq.contains", + indexOf: "seq.indexOf", + slice: "seq.slice", + concat: "seq.concat", + join: "str.join", + reverse: "seq.reverse", + some: "seq.any", + every: "seq.all", + find: "seq.find", +}; + +const METHOD_HELP: Record = { + toUpperCase: "use `str.asciiUpper` on a proven-ASCII value; full Unicode case mapping differs across targets", + toLowerCase: "use `str.asciiLower` on a proven-ASCII value", + replace: "write the replacement in source; host replace semantics differ", + replaceAll: "write the replacement in source; host replace semantics differ", + normalize: "Unicode normalization depends on the host's UCD version and is outside the subset", + localeCompare: "use `str.compare`, which is scalar order everywhere", + toFixed: "use `dec.fromFloat` and format in source, so rounding is explicit", + sort: "use `seq.sortStable` or `seq.sortStableBy`; host sorts differ on stability and on string order", + toString: "convert explicitly, for example with `str.fromInt`", +}; + +export type CheckMetrics = { + /** Loops whose accumulator ranges had to be widened instead of proven exactly. */ + widenedLoops: number; + /** Assignments whose range was clamped back into the platform-safe domain after widening. */ + clampedRanges: string[]; + /** Integer types that leave the platform-safe domain, which forces a wide representation. */ + wideIntegers: string[]; +}; + +type Binding = { + readonly name: string; + /** The type the binding was declared with; assignments are checked against this. */ + declared: SemType; + /** The flow-sensitive type at this point. */ + type: SemType; + readonly mutable: boolean; + /** True when flow analysis has proven an Option present, so uses unwrap it. */ + unwrapped: boolean; + /** True when a loop widened this binding's range instead of proving it exactly. */ + widened?: boolean; +}; + +type Narrowing = Map; + +type FuncSig = { + readonly qualified: string; + readonly params: readonly { name: string; type: SemType }[]; + readonly ret: SemType; + /** + * Which parameters were written as a bare `number`, aligned by index with `params`. Only an + * eagerly-checked function (a utility) acts on this — see "Inferring a bare `number`" in + * `checkFunction` — a library helper's `number` parameter is already the platform-safe default + * `Int` is, and specialization narrows it per call site exactly the same way. + */ + readonly inferParams: readonly boolean[]; + /** Whether the return type was written as a bare `number`: always inferred from the body. */ + readonly inferReturn: boolean; + readonly hir: HFunc; + readonly module: string; +}; + +/** A parameter or return type written as exactly `number`, with no type arguments. */ +function isBareNumber(type: HTypeExpr): boolean { + return type.kind === "ref" && type.name === "number" && type.args.length === 0; +} + +export function checkProgram( + modules: readonly HModule[], + diagnostics: Diagnostics, +): { program: CProgram; metrics: CheckMetrics } { + return new Checker(modules, diagnostics).run(); +} + +class Checker { + private readonly modules: readonly HModule[]; + private readonly diagnostics: Diagnostics; + private readonly records = new Map(); + private readonly enums = new Map(); + private readonly aliases = new Map(); + private readonly aliasSources = new Map(); + private readonly errors = new Map(); + private readonly functions = new Map(); + private readonly consts = new Map(); + private readonly checked = new Map(); + private readonly specializations = new Map(); + private readonly specializationCounts = new Map(); + private readonly moduleScopes = new Map>(); + private readonly metrics: CheckMetrics = { widenedLoops: 0, clampedRanges: [], wideIntegers: [] }; + private readonly entryPoints: string[] = []; + + constructor(modules: readonly HModule[], diagnostics: Diagnostics) { + this.modules = modules; + this.diagnostics = diagnostics; + } + + run(): { program: CProgram; metrics: CheckMetrics } { + this.collectDeclarations(); + this.resolveTypes(); + this.resolveSignatures(); + this.resolveConsts(); + const order = this.topologicalOrder(); + // Exported functions are the published contract, so they are checked with their declared + // signature. Internal helpers are checked per call site instead (see `instantiate`), which + // is what lets a helper like `digitAt` serve an 11 digit CPF and a 14 digit CNPJ without + // either losing the length proof its caller already has. + for (const qualified of order) { + const signature = this.functions.get(qualified); + if (signature === undefined || !isCheckedEagerly(signature)) continue; + this.checked.set(qualified, this.checkFunction(signature, signature.params, qualified)); + } + this.diagnostics.throwIfErrors(); + return { + program: { + records: this.records, + errors: this.errors, + functions: this.checked, + consts: this.consts, + entryPoints: this.entryPoints, + }, + metrics: this.metrics, + }; + } + + /* ---------------------------------------------------------------- * + * Declarations + * ---------------------------------------------------------------- */ + + private collectDeclarations(): void { + for (const record of BUILTIN_RECORDS) { + this.records.set(record.name, { + name: record.name, + fields: record.fields, + doc: record.doc, + exported: true, + }); + } + for (const name of BUILTIN_ERRORS) { + this.errors.set(name, { name, doc: "Raised by the engine's own intrinsics.", exported: true }); + } + + for (const module of this.modules) { + const scope = new Map(); + this.moduleScopes.set(module.path, scope); + for (const item of module.imports) { + const target = resolveImportPath(module.path, item.from); + for (const name of item.names) { + scope.set(name.local, `${target}::${name.imported}`); + } + } + for (const fn of module.functions) scope.set(fn.name, `${module.path}::${fn.name}`); + for (const constant of module.consts) scope.set(constant.name, `${module.path}::${constant.name}`); + } + + for (const module of this.modules) { + for (const decl of module.types) { + if (this.aliasSources.has(decl.name)) { + this.diagnostics.error( + "E_DUPLICATE_TYPE", + `the type \`${decl.name}\` is declared twice; type names are project-global`, + decl.span, + ); + continue; + } + this.aliasSources.set(decl.name, { type: decl.type, module: module.path }); + } + for (const decl of module.errors) { + this.errors.set(decl.name, { + name: decl.name, + base: decl.base === "DomainError" ? undefined : decl.base, + doc: decl.doc, + exported: decl.exported, + }); + } + } + } + + private resolveTypes(): void { + for (const [name, source] of this.aliasSources) { + const resolved = this.resolveTypeExpr(source.type, name); + this.aliases.set(name, resolved); + if (source.type.kind === "object") { + this.records.set(name, { + name, + fields: source.type.fields.map((field) => ({ + name: field.name, + type: this.resolveTypeExpr(field.type, `${name}.${field.name}`), + optional: field.optional, + doc: field.doc, + })), + doc: undefined, + exported: true, + }); + } + } + } + + private resolveTypeExpr(type: HTypeExpr, context?: string): SemType { + switch (type.kind) { + case "ref": + return this.resolveRef(type, context); + case "array": + return tList(this.resolveTypeExpr(type.elem, context)); + case "undefined": + return tOption(tNever); + case "literal": + return tEnum(context ?? "enum", [type.value]); + case "object": { + if (context === undefined) { + this.diagnostics.error( + "E_INLINE_RECORD", + "record types must be declared with a name", + type.span, + "declare `type Name = { … }` and refer to it; the core's records are nominal", + ); + return tNever; + } + return tRecord(context); + } + case "func": + return tLambda( + type.params.map((param) => this.resolveTypeExpr(param, context)), + this.resolveTypeExpr(type.ret, context), + ); + case "union": { + const options = type.options; + const hasUndefined = options.some((option) => option.kind === "undefined"); + const rest = options.filter((option) => option.kind !== "undefined"); + if (rest.every((option) => option.kind === "literal")) { + const members = rest.map((option) => (option as { value: string }).value); + const enumType = tEnum(context ?? "enum", members.sort()); + return hasUndefined ? tOption(enumType) : enumType; + } + if (rest.length === 1) { + const inner = this.resolveTypeExpr(rest[0]!, context); + return hasUndefined ? tOption(inner) : inner; + } + this.diagnostics.error( + "E_UNION", + "only string-literal unions and `T | undefined` are admitted so far", + type.span, + "a discriminated union of records needs the union admission rule; see docs/semantics.md", + ); + return tNever; + } + default: { + const exhaustive: never = type; + return exhaustive; + } + } + } + + private resolveRef( + type: Extract, + context?: string, + ): SemType { + const literalArg = (index: number): number | undefined => { + const arg = type.args[index]; + if (arg === undefined || arg.kind !== "literal") return undefined; + const parsed = Number(arg.value); + return Number.isFinite(parsed) ? parsed : undefined; + }; + + switch (type.name) { + case "Bool": + case "boolean": + return tBool; + case "Int": + return tIntDefault(); + case "number": + // A bare `number`. The platform-safe domain is only its *starting* point: for a + // parameter or a return type this is resolved through, `resolveSignatures` records + // that it needs inferring, and `checkFunction` replaces it with what the body actually + // proves (or refuses — see "Inferring a bare `number`" below). Everywhere else — a + // local, a record field, a list element, a lambda's own type — there is no call site + // and no exported contract to infer from, so it is simply the same unconstrained + // default `Int` already is; `let sum = 0` (no annotation) already infers the same way. + return tIntDefault(); + case "IntRange": { + const lo = literalArg(0); + const hi = literalArg(1); + if (lo === undefined || hi === undefined) { + this.diagnostics.error("E_INT_RANGE", "IntRange needs two integer literals", type.span); + return tIntDefault(); + } + return tInt(BigInt(lo), BigInt(hi)); + } + case "Float": + return tFloat; + case "Decimal": { + const scale = literalArg(0); + if (scale === undefined) { + this.diagnostics.error( + "E_DECIMAL_SCALE", + "Decimal needs a scale, for example `Decimal<2>`", + type.span, + ); + return tDecimal(2); + } + return tDecimal(scale); + } + case "String": + case "string": + return tString("none"); + case "Ascii": + return tString("ascii"); + case "Digits": + return tString("digits"); + case "AsciiOf": { + const length = literalArg(0); + if (length === undefined) { + this.diagnostics.error("E_LENGTH", "AsciiOf needs a length literal", type.span); + return tString("ascii"); + } + return tString("ascii", length, length); + } + case "DigitsOf": { + const length = literalArg(0); + if (length === undefined) { + this.diagnostics.error("E_LENGTH", "DigitsOf needs a length literal", type.span); + return tString("digits"); + } + return tString("digits", length, length); + } + case "CivilDate": + return tCivilDate; + case "Instant": + return tInstant; + case "Duration": + return tDuration; + case "Void": + case "void": + return tVoid; + case "List": { + const elem = type.args[0]; + if (elem === undefined) { + this.diagnostics.error("E_LIST", "List needs an element type", type.span); + return tList(tNever); + } + const min = literalArg(1) ?? 0; + const max = literalArg(2) ?? MAX_COLLECTION_LENGTH; + return tList(this.resolveTypeExpr(elem, context), min, max); + } + default: { + // A builtin record (HttpRequest and friends) has no alias source, but it is a + // nominal record type all the same. + if (this.records.has(type.name) && this.aliasSources.get(type.name) === undefined) { + return tRecord(type.name); + } + if (this.records.has(type.name) && this.aliasSources.get(type.name)?.type.kind === "object") { + return tRecord(type.name); + } + const alias = this.aliases.get(type.name); + if (alias !== undefined) return alias; + const source = this.aliasSources.get(type.name); + if (source !== undefined) { + if (source.type.kind === "object") return tRecord(type.name); + const resolved = this.resolveTypeExpr(source.type, type.name); + this.aliases.set(type.name, resolved); + return resolved; + } + this.diagnostics.error("E_UNKNOWN_TYPE", `unknown type \`${type.name}\``, type.span); + return tNever; + } + } + } + + private resolveSignatures(): void { + for (const module of this.modules) { + for (const fn of module.functions) { + const qualified = `${module.path}::${fn.name}`; + this.functions.set(qualified, { + qualified, + params: fn.params.map((param) => ({ + name: param.name, + type: this.resolveTypeExpr(param.type, `${fn.name}.${param.name}`), + })), + ret: this.resolveTypeExpr(fn.ret, fn.name), + inferParams: fn.params.map((param) => isBareNumber(param.type)), + inferReturn: isBareNumber(fn.ret), + hir: fn, + module: module.path, + }); + if (fn.exported && !module.path.includes("/")) this.entryPoints.push(qualified); + } + } + } + + private resolveConsts(): void { + for (const module of this.modules) { + for (const constant of module.consts) { + const qualified = `${module.path}::${constant.name}`; + if (constant.value.kind === "regex") { + try { + const regex = normalizeRegex(constant.value.source, constant.span); + if (constant.value.flags !== "") { + this.diagnostics.error( + "E_REGEX_FLAGS", + "regex flags are outside the subset: they mean different things in each dialect", + constant.span, + ); + } + this.consts.set(qualified, { + name: qualified, + type: tString("none"), + value: constant.value.source, + module: module.path, + regex, + }); + } catch (error) { + this.diagnostics.error( + "E_REGEX", + error instanceof RegexError ? error.message : String(error), + constant.span, + ); + } + continue; + } + const checker = new FunctionChecker(this, module.path, tVoid, new Map()); + const expected = + constant.declared === undefined ? undefined : this.resolveTypeExpr(constant.declared, constant.name); + const expr = checker.expr(constant.value, expected); + const value = tryEvalConst(expr); + if (value === undefined) { + this.diagnostics.error( + "E_NOT_CONSTANT", + `\`${constant.name}\` is not computable at compile time`, + constant.span, + "module level constants may only use literals, records, lists and pure intrinsics", + ); + continue; + } + this.consts.set(qualified, { + name: qualified, + type: expected ?? expr.type, + value, + module: module.path, + }); + } + } + } + + /* ---------------------------------------------------------------- * + * Call graph + * ---------------------------------------------------------------- */ + + private topologicalOrder(): string[] { + const edges = new Map>(); + for (const [qualified, signature] of this.functions) { + edges.set(qualified, new Set(this.callsOf(signature))); + } + const order: string[] = []; + const state = new Map(); + const visit = (name: string, stack: string[]): void => { + const status = state.get(name); + if (status === "done") return; + if (status === "visiting") { + const signature = this.functions.get(name); + this.diagnostics.error( + "E_RECURSION", + `recursion is outside the subset in this phase: ${[...stack, name].join(" -> ")}`, + signature?.hir.span ?? NO_SPAN, + "rewrite with a counted loop and an explicit stack", + ); + return; + } + state.set(name, "visiting"); + for (const callee of edges.get(name) ?? []) visit(callee, [...stack, name]); + state.set(name, "done"); + order.push(name); + }; + for (const name of this.functions.keys()) visit(name, []); + return order; + } + + private callsOf(signature: FuncSig): string[] { + const scope = this.moduleScopes.get(signature.module) ?? new Map(); + const found = new Set(); + const visitExpr = (expr: HExpr): void => { + switch (expr.kind) { + case "name": { + const target = scope.get(expr.name); + if (target !== undefined && this.functions.has(target)) found.add(target); + return; + } + case "member": + visitExpr(expr.target); + return; + case "index": + visitExpr(expr.target); + visitExpr(expr.index); + return; + case "call": + visitExpr(expr.callee); + expr.args.forEach(visitExpr); + return; + case "new": + expr.args.forEach(visitExpr); + return; + case "binary": + case "logical": + visitExpr(expr.left); + visitExpr(expr.right); + return; + case "unary": + visitExpr(expr.operand); + return; + case "ternary": + visitExpr(expr.test); + visitExpr(expr.then); + visitExpr(expr.otherwise); + return; + case "template": + for (const part of expr.parts) if (part.kind === "expr") visitExpr(part.expr); + return; + case "object": + for (const field of expr.fields) visitExpr(field.value); + return; + case "array": + expr.items.forEach(visitExpr); + return; + case "lambda": + expr.body.forEach(visitStmt); + return; + default: + return; + } + }; + const visitStmt = (statement: HStmt): void => { + switch (statement.kind) { + case "let": + visitExpr(statement.init); + return; + case "assign": + visitExpr(statement.target); + visitExpr(statement.value); + return; + case "if": + visitExpr(statement.test); + statement.then.forEach(visitStmt); + statement.otherwise?.forEach(visitStmt); + return; + case "switch": + visitExpr(statement.subject); + for (const entry of statement.cases) entry.body.forEach(visitStmt); + return; + case "forCounted": + visitExpr(statement.from); + visitExpr(statement.to); + statement.body.forEach(visitStmt); + return; + case "forOf": + visitExpr(statement.iterable); + statement.body.forEach(visitStmt); + return; + case "return": + if (statement.value !== undefined) visitExpr(statement.value); + return; + case "throw": + statement.args.forEach(visitExpr); + return; + case "expr": + visitExpr(statement.expr); + return; + case "block": + statement.body.forEach(visitStmt); + return; + default: + return; + } + }; + signature.hir.body.forEach(visitStmt); + return [...found]; + } + + /** + * Checks a function against the argument types a call site really has. + * + * Exported functions always use their declared signature. An internal helper is specialized: + * the refinements its caller proved (a length, a character class, a range) flow into the + * helper's body, so the proof composes across function boundaries instead of stopping at them. + * Specializations that coincide are shared, and `MAX_SPECIALIZATIONS` caps the fan-out. + */ + instantiate( + qualified: string, + argTypes: readonly SemType[], + ): { name: string; ret: SemType; effects: EffectSet; params: readonly { name: string; type: SemType }[] } | undefined { + const signature = this.functions.get(qualified); + if (signature === undefined) return undefined; + if (isCheckedEagerly(signature)) { + const checked = this.checked.get(qualified); + return { + name: qualified, + ret: signature.ret, + effects: checked?.effects ?? PURE, + params: signature.params, + }; + } + const params = signature.params.map((param, index) => { + const actual = argTypes[index]; + return { + name: param.name, + type: actual !== undefined && isSubtype(actual, param.type) ? actual : param.type, + }; + }); + const key = `${qualified}(${params.map((param) => typeToString(param.type)).join(", ")})`; + const existing = this.specializations.get(key); + if (existing !== undefined) { + const checked = this.checked.get(existing)!; + return { name: existing, ret: checked.ret, effects: checked.effects, params }; + } + const count = this.specializationCounts.get(qualified) ?? 0; + if (count >= MAX_SPECIALIZATIONS) { + const fallback = this.checked.get(qualified); + if (fallback !== undefined) { + return { + name: qualified, + ret: fallback.ret, + effects: fallback.effects, + params: signature.params, + }; + } + } + const name = count === 0 ? qualified : `${qualified}$${count}`; + this.specializationCounts.set(qualified, count + 1); + this.specializations.set(key, name); + // Reserve the slot before checking, so a call made while this body is being checked sees + // the name rather than starting a second, identical specialization. + const checkedFunction = this.checkFunction(signature, params, name); + this.checked.set(name, checkedFunction); + return { name, ret: checkedFunction.ret, effects: checkedFunction.effects, params }; + } + + private checkFunction( + signature: FuncSig, + paramTypes: readonly { name: string; type: SemType }[], + name: string, + ): CFunc { + const scope = new Map(); + for (const param of paramTypes) { + scope.set(param.name, { + name: param.name, + declared: param.type, + type: param.type, + mutable: false, + unwrapped: false, + }); + } + const checker = new FunctionChecker(this, signature.module, signature.ret, scope); + // Inferring a bare `number`: a library helper's parameter is specialized per call site + // regardless (see `instantiate`, and ADR 0004), so it needs nothing extra here — it starts + // and stays the platform-safe default, exactly like a declared `Int`, until a caller's own + // proven type is substituted in. A utility has no call site to take a range from: its + // contract has to come from its own body, so any bare-`number` parameter is tracked while + // the body is checked, and `provenParamType` below turns what was tracked into either the + // narrower published type a guard proved, or a refusal. + const inferredParams = isCheckedEagerly(signature) + ? signature.hir.params + .filter((_, index) => signature.inferParams[index] === true) + .map((param) => param.name) + : []; + if (inferredParams.length > 0) checker.beginParamInference(inferredParams); + const body = checker.block(signature.hir.body); + // A specialization may prove a narrower result than the declaration promises, and its + // callers should see that proof; a public signature stays exactly as declared — unless the + // declaration itself was a bare `number`, which never states a ceiling to stay under: its + // published return type is always what the body computes. + const inferred = returnTypeOf(body); + const ret = signature.inferReturn + ? inferred.kind !== "Never" + ? inferred + : signature.ret + : !isCheckedEagerly(signature) && inferred.kind !== "Never" && isSubtype(inferred, signature.ret) + ? inferred + : signature.ret; + if (signature.ret.kind !== "Void" && !checker.alwaysReturns(body)) { + this.diagnostics.error( + "E_MISSING_RETURN", + `\`${signature.hir.name}\` does not return on every path`, + signature.hir.span, + ); + } + const params = paramTypes.map((param, index) => { + if (!inferredParams.includes(param.name)) { + return { name: param.name, type: param.type, doc: signature.hir.params[index]?.name }; + } + const proven = checker.provenParamType(param.name); + const span = signature.hir.params[index]?.span ?? signature.hir.span; + if (proven === undefined) { + this.diagnostics.error( + "E_BARE_NUMBER", + `\`${param.name}\` is never used, so \`${signature.hir.name}\` proves nothing about its range`, + span, + `remove the parameter, or write \`${param.name}: Int\` if the full range really is what is meant`, + ); + } else if (isUnconstrainedInt(proven)) { + this.diagnostics.error( + "E_BARE_NUMBER", + `\`${param.name}\` is a bare \`number\`, and \`${signature.hir.name}\` never narrows it before using it`, + span, + `add a guard before ${param.name} is used, for example ` + + `\`if (${param.name} < 0 || ${param.name} > 99) return …;\` — the range a guard like that ` + + `proves becomes ${param.name}'s published contract; write \`${param.name}: Int\` instead if ` + + `the full range really is what is meant`, + ); + } + return { name: param.name, type: proven ?? param.type, doc: signature.hir.params[index]?.name }; + }); + return { + name, + module: signature.module, + localName: name === signature.qualified ? signature.hir.name : name.split("::")[1]!, + params, + ret, + effects: checker.effects, + body, + // "Exported" means part of the published core API: a utility, not a library helper. + exported: isEntryPoint(signature) && name === signature.qualified, + // The source's own `export` keyword, independent of utility-ness: true for any exported, + // unspecialized function, root module or `lib/`. Every utility is module-exported too + // (a utility's own `export` is what makes it one), but the reverse does not hold — a + // `lib/` function's `export` only ever makes it reachable across generated modules, never + // a utility, since `isEntryPoint` excludes anything under a subdirectory. + moduleExported: signature.hir.exported && name === signature.qualified, + doc: signature.hir.doc, + span: signature.hir.span, + // The call graph of the Core, not of the source: after specialization a call site + // names the specialization it was checked against. + calls: collectCalls(body), + usesEnv: false, + }; + } + + /* ---------------------------------------------------------------- * + * Shared lookups used by FunctionChecker + * ---------------------------------------------------------------- */ + + get diags(): Diagnostics { + return this.diagnostics; + } + + get recordTable(): Map { + return this.records; + } + + get errorTable(): Map { + return this.errors; + } + + get constTable(): Map { + return this.consts; + } + + get metricTable(): CheckMetrics { + return this.metrics; + } + + scopeOf(module: string): Map { + return this.moduleScopes.get(module) ?? new Map(); + } + + signatureOf(qualified: string): FuncSig | undefined { + return this.functions.get(qualified); + } + + effectsOf(qualified: string): EffectSet { + return this.checked.get(qualified)?.effects ?? PURE; + } + + /** The checked body of a function, for the analyses that have to look through a call. */ + bodyOf(qualified: string): readonly CStmt[] | undefined { + return this.checked.get(qualified)?.body; + } + + resolveType(type: HTypeExpr, context?: string): SemType { + return this.resolveTypeExpr(type, context); + } + + typeOfAlias(name: string): SemType | undefined { + return this.aliases.get(name); + } +} + +/** + * A utility is an exported function of a module at the source root; everything under a + * subdirectory (`lib/…`) is library code. Utilities are the published core API, so they keep + * their declared signature; library code is specialized per call site. + */ +function isEntryPoint(signature: FuncSig): boolean { + return signature.hir.exported && !signature.module.includes("/"); +} + +/** + * Functions checked against their declared signature rather than per call site: the project's + * utilities, and the engine's own standard library, whose functions a portable lowering may call + * without any source in the project ever naming them. + */ +function isCheckedEagerly(signature: FuncSig): boolean { + return isEntryPoint(signature) || STDLIB_SURFACE.includes(signature.qualified); +} + +/** Whether a statement can reach a `break` that belongs to the switch rather than to a loop. */ +function escapesCase(statement: HStmt): boolean { + switch (statement.kind) { + case "break": + return true; + case "if": + return statement.then.some(escapesCase) || (statement.otherwise ?? []).some(escapesCase); + case "block": + return statement.body.some(escapesCase); + case "switch": + // A nested switch owns its own breaks. + return false; + case "forCounted": + case "forOf": + // A loop inside the case owns its own breaks. + return false; + default: + return false; + } +} + +/** Every operation a Core body performs, including inside nested lambdas. */ +function operationsOf(body: readonly CStmt[]): Extract[] { + const found: Extract[] = []; + const expr = (node: CExpr): void => { + switch (node.kind) { + case "op": + found.push(node); + node.args.forEach(expr); + return; + case "call": + node.args.forEach(expr); + return; + case "record": + node.fields.forEach((field) => expr(field.value)); + return; + case "list": + node.items.forEach(expr); + return; + case "field": + expr(node.target); + return; + case "some": + expr(node.inner); + return; + case "cond": + expr(node.test); + expr(node.then); + expr(node.otherwise); + return; + case "and": + case "or": + expr(node.left); + expr(node.right); + return; + case "not": + expr(node.operand); + return; + case "lambda": + found.push(...operationsOf(node.body)); + return; + default: + return; + } + }; + const statement = (node: CStmt): void => { + switch (node.kind) { + case "let": + expr(node.init); + return; + case "assign": + case "push": + expr(node.value); + return; + case "setIndex": + expr(node.index); + expr(node.value); + return; + case "if": + expr(node.test); + node.then.forEach(statement); + node.otherwise.forEach(statement); + return; + case "switch": + expr(node.subject); + node.cases.forEach((entry) => entry.body.forEach(statement)); + node.otherwise?.forEach(statement); + return; + case "forRange": + expr(node.from); + expr(node.to); + node.body.forEach(statement); + return; + case "forEach": + expr(node.iterable); + node.body.forEach(statement); + return; + case "return": + if (node.value !== undefined) expr(node.value); + return; + case "fail": + node.args.forEach(expr); + return; + case "expr": + expr(node.expr); + return; + default: + return; + } + }; + body.forEach(statement); + return found; +} + +/** Every function a Core body calls, fully qualified and de-duplicated. */ +const callsOf = (body: readonly CStmt[]): string[] => collectCalls(body); + +function collectCalls(body: readonly CStmt[]): string[] { + const found = new Set(); + const expr = (node: CExpr): void => { + switch (node.kind) { + case "call": + found.add(node.fn); + node.args.forEach(expr); + return; + case "op": + node.args.forEach(expr); + return; + case "record": + node.fields.forEach((field) => expr(field.value)); + return; + case "list": + node.items.forEach(expr); + return; + case "field": + expr(node.target); + return; + case "some": + expr(node.inner); + return; + case "cond": + expr(node.test); + expr(node.then); + expr(node.otherwise); + return; + case "and": + case "or": + expr(node.left); + expr(node.right); + return; + case "not": + expr(node.operand); + return; + case "lambda": + node.body.forEach(statement); + return; + default: + return; + } + }; + const statement = (node: CStmt): void => { + switch (node.kind) { + case "let": + expr(node.init); + return; + case "assign": + expr(node.value); + return; + case "setIndex": + expr(node.index); + expr(node.value); + return; + case "push": + expr(node.value); + return; + case "if": + expr(node.test); + node.then.forEach(statement); + node.otherwise.forEach(statement); + return; + case "switch": + expr(node.subject); + node.cases.forEach((entry) => entry.body.forEach(statement)); + node.otherwise?.forEach(statement); + return; + case "forRange": + expr(node.from); + expr(node.to); + node.body.forEach(statement); + return; + case "forEach": + expr(node.iterable); + node.body.forEach(statement); + return; + case "return": + if (node.value !== undefined) expr(node.value); + return; + case "fail": + node.args.forEach(expr); + return; + case "expr": + expr(node.expr); + return; + default: + return; + } + }; + body.forEach(statement); + return [...found]; +} + +/** + * The character class a pattern denotes, when it is exactly one class. + * + * `re.retain` is defined only for that shape: a class is linear, allocation-bounded and means the + * same thing in every engine, and it is also what tells the checker how to refine the result. + */ +function singleClassOf(regex: NormalizedRegex): "ascii" | "digits" | undefined { + const node = regex.node; + if (node.kind !== "class") return undefined; + const digits = node.ranges.every((range) => range.lo >= 0x30 && range.hi <= 0x39); + if (digits) return "digits"; + return node.ranges.every((range) => range.hi < 0x80) ? "ascii" : undefined; +} + +/** + * Whether `index` is proven inside `[0, subject.min)`, so a positional read of `subject` can + * never miss: this is the same test `str.charAt`, `str.codeAt` and `seq.get` each make in their + * own signature, duplicated here so the idiomatic `s[i]`/`xs[i]` sugar can choose between the + * unchecked accessor and its checked, Option-returning form *before* calling either one. + */ +function provablyInRange( + index: CExpr, + subject: Extract | Extract, +): boolean { + return index.type.kind === "Int" && index.type.lo >= 0n && index.type.hi < BigInt(subject.min); +} + +/** + * The inner text of `source` when it is, in full, a single negated character class such as + * `[^0-9]` — the shape `value.replace(/[^0-9]/g, "")` needs to become `re.retain` on the class's + * own (un-negated) content. `undefined` when `source` is anything else: a partial pattern, an + * escaped `]` as the very first character (rare enough not to special-case), or not a class at + * all. Text surgery rather than parsing the class and re-emitting it, because the *bracketed* + * content has to survive unchanged for `normalizeRegex` to parse on the retain side. + */ +function negatedClassInner(source: string): string | undefined { + if (!source.startsWith("[^")) return undefined; + let cursor = 2; + while (cursor < source.length && source[cursor] !== "]") { + cursor += source[cursor] === "\\" ? 2 : 1; + } + return cursor === source.length - 1 && source[cursor] === "]" ? source.slice(2, cursor) : undefined; +} + +function resolveImportPath(from: string, specifier: string): string { + if (!specifier.startsWith(".")) return specifier; + const base = from.split("/").slice(0, -1); + const parts = specifier.replace(/\.ts$/, "").split("/"); + for (const part of parts) { + if (part === ".") continue; + if (part === "..") base.pop(); + else base.push(part); + } + return base.join("/"); +} + +/* ==================================================================== * + * Function level checking + * ==================================================================== */ + +/** How many times a loop body is re-checked before ranges are widened instead of proven. */ +const MAX_LOOP_ITERATIONS = 64; + +/** How many differently refined versions of one internal helper the checker will produce. */ +const MAX_SPECIALIZATIONS = 16; + +/** + * The largest single step a widened loop counter may take and still be clamped back into the + * platform-safe domain. A loop runs at most MAX_COLLECTION_LENGTH (2^31 - 1) times, so a step of + * at most 2^22 keeps the value inside 2^53. + */ +const MAX_WIDENED_STEP = 2n ** 22n; + +type ScopeSnapshot = Map; + +class FunctionChecker { + private readonly checker: Checker; + private readonly module: string; + private readonly returnType: SemType; + private readonly scope: Map; + private quiet = false; + /** + * The states `break` and `continue` leave the innermost loop's body with. + * + * Neither leaves the function, so an assignment made just before one is live: a `continue`'s + * state seeds the next iteration, and a `break`'s state seeds the code after the loop. Both + * are collected here because the statement that produced them is not on the path that falls + * out of the body, which is the only path the straight-line walk sees. + */ + private breakStates: ScopeSnapshot[] | undefined; + private continueStates: ScopeSnapshot[] | undefined; + effects: EffectSet = PURE; + /** + * The type observed at each read of a bare-`number` parameter that is being inferred (see + * `beginParamInference`), joined across every read that counts. `undefined` for a name means it + * has never been read yet; a name absent from the map is not being inferred at all. Shared with + * any lambda checked inside this function's body (`lambda()` passes the same map along), so a + * closure that reads the parameter contributes to the same inference the enclosing body does. + */ + private paramObservations: Map | undefined; + /** + * True while checking the test of an `if` or a ternary — the one expression position whose + * *result* narrows a binding rather than *using* its value, so a bare-`number` parameter read + * only there proves nothing about the range the rest of the function relies on. Nested inside + * another condition's test, a value position (a ternary's `then`/`otherwise`) still reads as + * "inside a condition" for as long as the enclosing test has not finished, which is exactly the + * scoping `withCondition`'s save/restore gives it. + */ + private inCondition = false; + + constructor( + checker: Checker, + module: string, + returnType: SemType, + scope: Map, + ) { + this.checker = checker; + this.module = module; + this.returnType = returnType; + this.scope = scope; + } + + private report(code: string, message: string, span: Span, suggestion?: string): void { + if (!this.quiet) this.checker.diags.error(code, message, span, suggestion); + } + + private addEffects(set: EffectSet): void { + this.effects = unionEffects(this.effects, set); + } + + /* ---------------------------------------------------------------- * + * Inferring a bare `number` parameter from the guards a utility writes + * ---------------------------------------------------------------- */ + + /** Starts tracking reads of these parameter names, for `provenParamType` once the body is checked. */ + beginParamInference(names: readonly string[]): void { + this.paramObservations = new Map(names.map((name) => [name, undefined])); + } + + /** The range proven for a tracked parameter — `undefined` when it was never read at all. */ + provenParamType(name: string): SemType | undefined { + return this.paramObservations?.get(name); + } + + /** Records a read of `name` for inference, unless it is inside a condition's own test. */ + private noteRead(name: string, type: SemType): void { + if (this.inCondition || this.paramObservations === undefined) return; + if (!this.paramObservations.has(name)) return; + const current = this.paramObservations.get(name); + this.paramObservations.set(name, current === undefined ? type : join(current, type)); + } + + /** Runs `evaluate` with `inCondition` set, restoring the previous value even if it was already set. */ + private withCondition(evaluate: () => T): T { + const outer = this.inCondition; + this.inCondition = true; + try { + return evaluate(); + } finally { + this.inCondition = outer; + } + } + + /* ---------------------------------------------------------------- * + * Scope plumbing + * ---------------------------------------------------------------- */ + + private snapshot(): ScopeSnapshot { + const snapshot: ScopeSnapshot = new Map(); + for (const [name, binding] of this.scope) { + snapshot.set(name, { type: binding.type, unwrapped: binding.unwrapped }); + } + return snapshot; + } + + private restore(snapshot: ScopeSnapshot): void { + for (const [name, state] of snapshot) { + const binding = this.scope.get(name); + if (binding !== undefined) { + binding.type = state.type; + binding.unwrapped = state.unwrapped; + } + } + for (const name of [...this.scope.keys()]) { + if (!snapshot.has(name)) this.scope.delete(name); + } + } + + private mergeInto(left: ScopeSnapshot, right: ScopeSnapshot): void { + for (const [name, binding] of this.scope) { + const a = left.get(name); + const b = right.get(name); + if (a === undefined || b === undefined) continue; + binding.type = join(a.type, b.type); + binding.unwrapped = a.unwrapped && b.unwrapped; + } + } + + /** Sets every binding to the join of the states it has across `states`. */ + private applyJoin(states: readonly ScopeSnapshot[]): void { + for (const [name, binding] of this.scope) { + let merged: SemType | undefined; + let unwrapped = true; + for (const state of states) { + const entry = state.get(name); + if (entry === undefined) continue; + merged = merged === undefined ? entry.type : join(merged, entry.type); + unwrapped &&= entry.unwrapped; + } + if (merged === undefined) continue; + binding.type = merged; + binding.unwrapped = unwrapped; + } + } + + private applyNarrowing(narrowing: Narrowing): void { + for (const [name, type] of narrowing) { + const binding = this.scope.get(name); + if (binding === undefined) continue; + binding.type = type; + binding.unwrapped = binding.declared.kind === "Option" && type.kind !== "Option"; + } + } + + /* ---------------------------------------------------------------- * + * Statements + * ---------------------------------------------------------------- */ + + block(body: readonly HStmt[]): CStmt[] { + const result: CStmt[] = []; + for (const statement of body) result.push(...this.stmt(statement)); + return result; + } + + /** True when the body returns (or fails) on every path; used for return checking. */ + alwaysReturns(body: readonly CStmt[]): boolean { + for (const statement of body) { + if (statement.kind === "return" || statement.kind === "fail") return true; + if ( + statement.kind === "if" && + statement.otherwise.length > 0 && + this.alwaysReturns(statement.then) && + this.alwaysReturns(statement.otherwise) + ) { + return true; + } + if (statement.kind === "switch") { + const complete = + statement.otherwise !== undefined && + statement.cases.every((entry) => this.alwaysReturns(entry.body)) && + this.alwaysReturns(statement.otherwise); + if (complete) return true; + } + } + return false; + } + + stmt(statement: HStmt): CStmt[] { + switch (statement.kind) { + case "let": { + const declared = + statement.declared === undefined + ? undefined + : this.checker.resolveType(statement.declared, statement.name); + const init = this.expr(statement.init, declared); + if (declared !== undefined && !isSubtype(init.type, declared)) { + this.report( + "E_TYPE", + `cannot assign ${typeToString(init.type)} to ${typeToString(declared)}`, + statement.span, + ); + } + const type = declared ?? init.type; + this.scope.set(statement.name, { + name: statement.name, + declared: type, + type: init.type, + mutable: statement.mutable, + unwrapped: false, + }); + this.noteWideInteger(type, statement.span); + return [ + { kind: "let", name: statement.name, mutable: statement.mutable, init, type, span: statement.span }, + ]; + } + case "assign": + return this.assignment(statement); + case "if": { + const test = this.withCondition(() => this.expr(statement.test, tBool)); + this.requireBool(test, statement.span); + const { whenTrue, whenFalse } = deriveNarrowing(test); + const entry = this.snapshot(); + this.applyNarrowing(whenTrue); + const then = this.block(statement.then); + const afterThen = this.snapshot(); + this.restore(entry); + this.applyNarrowing(whenFalse); + const otherwise = statement.otherwise === undefined ? [] : this.block(statement.otherwise); + const afterElse = this.snapshot(); + + const thenExits = this.alwaysReturns(then) || exitsScope(then); + const elseExits = this.alwaysReturns(otherwise) || exitsScope(otherwise); + if (thenExits && !elseExits) { + // `if (!ok) return …` leaves the negated fact in force for the rest of the block. + this.restore(afterElse); + } else if (elseExits && !thenExits) { + this.restore(afterThen); + } else { + this.restore(entry); + this.mergeInto(afterThen, afterElse); + } + return [{ kind: "if", test, then, otherwise, span: statement.span }]; + } + case "switch": + return this.switchStatement(statement); + case "forCounted": + return this.countedLoop(statement); + case "forOf": + return this.forEachLoop(statement); + case "return": { + if (statement.value === undefined) { + return [{ kind: "return", span: statement.span }]; + } + const checked = this.expr(statement.value, this.returnType); + const value = + this.returnType.kind === "Option" && checked.type.kind !== "Option" && checked.type.kind !== "Never" + ? ({ kind: "some", inner: checked, type: tOption(checked.type), span: checked.span } as CExpr) + : checked; + // A lambda passed to a combinator has no declared return type: the combinator's own + // signature checks the result, so there is nothing to compare against here. + if ( + this.returnType.kind !== "Never" && + !isSubtype(value.type, this.returnType) && + !fitsPlatformDomain(value.type, this.returnType) + ) { + this.report( + "E_RETURN_TYPE", + `returning ${typeToString(value.type)} where ${typeToString(this.returnType)} is declared`, + statement.span, + ); + } + return [{ kind: "return", value, span: statement.span }]; + } + case "throw": { + if (!this.checker.errorTable.has(statement.errorClass)) { + this.report( + "E_UNKNOWN_ERROR", + `unknown error type \`${statement.errorClass}\``, + statement.span, + "declare it in source: `export class MyError extends DomainError {}`", + ); + } + this.effects = withFail(this.effects, statement.errorClass); + const args = statement.args.map((arg) => this.expr(arg, tString("none"))); + return [{ kind: "fail", errorClass: statement.errorClass, args, span: statement.span }]; + } + case "break": + this.breakStates?.push(this.snapshot()); + return [{ kind: "break", span: statement.span }]; + case "continue": + this.continueStates?.push(this.snapshot()); + return [{ kind: "continue", span: statement.span }]; + case "block": + return this.block(statement.body); + case "expr": + return this.expressionStatement(statement.expr, statement.span); + default: { + const exhaustive: never = statement; + return exhaustive; + } + } + } + + private expressionStatement(expr: HExpr, span: Span): CStmt[] { + // `xs.push(v)` is the one mutation the subset allows, and only on a local list. + if (expr.kind === "call" && expr.callee.kind === "member" && expr.callee.name === "push") { + const target = expr.callee.target; + if (target.kind !== "name") { + this.report("E_PUSH_TARGET", "push is allowed only on a local list", span); + return []; + } + const binding = this.scope.get(target.name); + if (binding === undefined || binding.type.kind !== "List") { + this.report("E_PUSH_TARGET", `\`${target.name}\` is not a local list`, span); + return []; + } + if (!binding.mutable) { + this.report( + "E_PUSH_CONST", + `\`${target.name}\` is not mutable`, + span, + "declare the list with `let` while it is being built", + ); + } + const value = this.expr(expr.args[0]!, binding.type.elem); + if (!isSubtype(value.type, binding.type.elem)) { + this.report( + "E_TYPE", + `cannot push ${typeToString(value.type)} into ${typeToString(binding.type)}`, + span, + ); + } + const listType = binding.type; + binding.type = tList( + listType.elem, + Math.min(listType.min + 1, MAX_COLLECTION_LENGTH), + Math.min(listType.max + 1, MAX_COLLECTION_LENGTH), + ); + binding.declared = tList(listType.elem, 0, MAX_COLLECTION_LENGTH); + return [{ kind: "push", name: target.name, value, span }]; + } + const checked = this.expr(expr); + return [{ kind: "expr", expr: checked, span }]; + } + + private assignment(statement: Extract): CStmt[] { + const target = statement.target; + if (target.kind === "index") { + if (target.target.kind !== "name") { + this.report("E_ASSIGN_TARGET", "only a local list element may be assigned", statement.span); + return []; + } + const binding = this.scope.get(target.target.name); + if (binding === undefined || binding.type.kind !== "List" || !binding.mutable) { + this.report( + "E_ASSIGN_TARGET", + `\`${target.target.name}\` is not a mutable local list`, + statement.span, + ); + return []; + } + const index = this.expr(target.index, tIntDefault()); + if (index.type.kind === "Int" && (index.type.lo < 0n || index.type.hi >= BigInt(binding.type.min))) { + this.report( + "E_INDEX", + `the index may be out of range (${typeToString(index.type)} into ${typeToString(binding.type)})`, + statement.span, + "guard the index against the list length", + ); + } + const value = this.expr(target.index === undefined ? statement.value : statement.value, binding.type.elem); + return [{ kind: "setIndex", name: target.target.name, index, value, span: statement.span }]; + } + if (target.kind !== "name") { + this.report( + "E_ASSIGN_TARGET", + "only locals are mutable", + statement.span, + "records are immutable; build a new one instead", + ); + return []; + } + const binding = this.scope.get(target.name); + if (binding === undefined) { + this.report("E_UNKNOWN_NAME", `unknown name \`${target.name}\``, statement.span); + return []; + } + if (!binding.mutable) { + this.report("E_IMMUTABLE", `\`${target.name}\` is not mutable`, statement.span, "declare it with `let`"); + } + const current: CExpr = { + kind: "local", + name: binding.name, + type: binding.type, + span: statement.span, + }; + const raw = this.expr(statement.value, binding.declared); + let value = raw; + if (statement.op !== "=") { + const op = statement.op === "+=" ? "+" : statement.op === "-=" ? "-" : "*"; + value = this.binaryOp(op, current, raw, statement.span); + } + let assigned = value.type; + if (!isSubtype(assigned, widenAssignment(binding.declared))) { + // A loop-carried counter whose exact range the analysis could not compute is widened to + // the platform-safe domain. Clamping back into it is sound only while one step is small: + // every loop's trip count is bounded by MAX_COLLECTION_LENGTH, so a bounded step cannot + // leave the domain. A step that is not bounded is still an error. + const step = stepSize(binding.declared, assigned); + const platformDomain = + binding.declared.kind === "Int" && binding.declared.hi >= SAFE_INT_HI - MAX_WIDENED_STEP; + if ((binding.widened === true || platformDomain) && step !== undefined && step <= MAX_WIDENED_STEP) { + this.checker.metricTable.clampedRanges.push( + `${statement.span.file}:${statement.span.start} ${typeToString(assigned)} clamped to ${typeToString(binding.declared)}`, + ); + assigned = binding.declared; + } else { + this.report( + "E_UNPROVEN_RANGE", + `cannot prove ${typeToString(value.type)} stays within ${typeToString(binding.declared)}`, + statement.span, + "annotate the binding with a wider `IntRange`, or narrow the operands", + ); + } + } + binding.type = assigned; + binding.unwrapped = false; + return [{ kind: "assign", name: target.name, value, span: statement.span }]; + } + + private switchStatement(statement: Extract): CStmt[] { + const subject = this.expr(statement.subject); + if (subject.type.kind !== "Enum") { + this.report( + "E_SWITCH_SUBJECT", + `switch works over a string-literal union, not ${typeToString(subject.type)}`, + statement.span, + "use `if`/`else` for other types", + ); + } + const entry = this.snapshot(); + const cases: { values: Value[]; body: CStmt[] }[] = []; + let otherwise: CStmt[] | undefined; + const covered = new Set(); + // Only a case that falls off its own end reaches the code after the switch; one that always + // returns or leaves the enclosing loop (`exitsScope`, mirroring how `if`/`else` tells the two + // apart) never gets there, so its ending state is not part of what "after the switch" means. + const fallThroughExits: ScopeSnapshot[] = []; + const collect = (body: readonly CStmt[]): void => { + if (!(this.alwaysReturns(body) || exitsScope(body))) fallThroughExits.push(this.snapshot()); + }; + for (const caseNode of statement.cases) { + this.restore(entry); + const caseBody = this.caseBody(caseNode.body); + if (caseNode.test === undefined) { + otherwise = this.block(caseBody); + collect(otherwise); + continue; + } + const test = this.expr(caseNode.test, subject.type); + const value = tryEvalConst(test); + if (value === undefined || typeof value !== "string") { + this.report("E_SWITCH_CASE", "case labels must be string literals", caseNode.span); + continue; + } + covered.add(value); + if (subject.kind === "local") { + const binding = this.scope.get(subject.name); + if (binding !== undefined && binding.type.kind === "Enum") { + binding.type = tEnum(binding.type.name, [value]); + } + } + const body = this.block(caseBody); + cases.push({ values: [value], body }); + collect(body); + } + this.restore(entry); + if (subject.type.kind === "Enum" && otherwise === undefined) { + const missing = subject.type.members.filter((member) => !covered.has(member)); + if (missing.length > 0) { + this.report( + "E_NON_EXHAUSTIVE", + `switch does not cover ${missing.map((member) => JSON.stringify(member)).join(", ")}`, + statement.span, + "add the missing cases, or a default", + ); + } + } + // Every case that can fall off its own end reaches this point, so what a mutable local holds + // after the switch has to cover every one of them — exactly the join a loop's fixpoint takes + // over `break`/`continue` states (`loopFixpoint`), for the same reason: an effect a branch + // produces is only sound to forget once every path that could carry it forward is accounted + // for, not merely the path the checker happened to look at last. + this.applyJoin(fallThroughExits); + return [{ kind: "switch", subject, cases, otherwise, span: statement.span }]; + } + + /** + * The statements of a switch case, without the `break` that ends it. + * + * In TypeScript that `break` leaves the *switch*; the Core's switch never falls through, so it + * carries no meaning and is dropped. It cannot be kept: a target that prints the switch as a + * chain of conditionals — Python does — would read it as leaving the enclosing loop. A `break` + * anywhere else in a case would mean the same thing and cannot be expressed, so it is refused + * rather than mistranslated. + */ + private caseBody(body: readonly HStmt[]): readonly HStmt[] { + const trimmed = body.length > 0 && body[body.length - 1]!.kind === "break" ? body.slice(0, -1) : body; + for (const statement of trimmed) { + if (escapesCase(statement)) { + this.report( + "E_SWITCH_BREAK", + "a `break` inside a switch case cannot leave the switch early", + statement.span, + "restructure the case so it ends at its last statement, or use `if`/`else`", + ); + } + } + return trimmed; + } + + /* ---------------------------------------------------------------- * + * Loops: the interval fixpoint lives here + * ---------------------------------------------------------------- */ + + private countedLoop(statement: Extract): CStmt[] { + const from = this.expr(statement.from, tIntDefault()); + const to = this.expr(statement.to, tIntDefault()); + if (from.type.kind !== "Int" || to.type.kind !== "Int") { + this.report("E_FOR_BOUNDS", "counted loops range over integers", statement.span); + return []; + } + const adjust = statement.inclusive ? 0n : 1n; + const counterType = + statement.step > 0n + ? tInt(from.type.lo, to.type.hi - adjust < from.type.lo ? from.type.lo : to.type.hi - adjust) + : tInt(to.type.lo + adjust > from.type.hi ? from.type.hi : to.type.lo + adjust, from.type.hi); + const trips = + statement.step > 0n + ? to.type.hi - adjust - from.type.lo + 1n + : from.type.hi - (to.type.lo + adjust) + 1n; + + const body = this.loopFixpoint(trips, () => { + this.scope.set(statement.name, { + name: statement.name, + declared: counterType, + type: counterType, + mutable: false, + unwrapped: false, + }); + return this.block(statement.body); + }); + return [ + { + kind: "forRange", + name: statement.name, + type: counterType, + from, + to, + inclusive: statement.inclusive, + step: statement.step, + body, + span: statement.span, + }, + ]; + } + + private forEachLoop(statement: Extract): CStmt[] { + const iterable = this.expr(statement.iterable); + let elem: SemType; + let trips: bigint; + if (iterable.type.kind === "List") { + elem = iterable.type.elem; + trips = BigInt(iterable.type.max); + } else if (iterable.type.kind === "String") { + this.report( + "E_ITERATE_STRING", + "iterate the scalars explicitly", + statement.span, + "use `str.codePoints(value)`, which every target iterates the same way", + ); + elem = tInt(0n, 0x10ffffn); + trips = BigInt(iterable.type.max); + } else { + this.report("E_ITERABLE", `${typeToString(iterable.type)} is not iterable`, statement.span); + return []; + } + + const body = this.loopFixpoint(trips, () => { + this.scope.set(statement.name, { + name: statement.name, + declared: elem, + type: elem, + mutable: false, + unwrapped: false, + }); + return this.block(statement.body); + }); + return [ + { + kind: "forEach", + name: statement.name, + type: elem, + iterable, + body, + span: statement.span, + }, + ]; + } + + /** + * Re-checks a loop body until the mutable locals' types stop growing. + * + * With a proven trip count of at most `MAX_LOOP_ITERATIONS`, the fixpoint is exact: the range + * of an accumulator is the one it really has on the last iteration. Above that, ranges widen + * to the platform-safe domain and the loop is counted in the metrics, because a widened range + * usually means the source should carry a tighter annotation. + */ + private loopFixpoint(trips: bigint, check: () => CStmt[]): CStmt[] { + const entry = this.snapshot(); + const bounded = trips <= BigInt(MAX_LOOP_ITERATIONS) && trips >= 0n; + // The loop head sees the state after at most `trips - 1` completed bodies, so that many + // merges are an exact fixpoint rather than an over-approximation. + const iterations = bounded ? Math.max(0, Number(trips) - 1) : MAX_LOOP_ITERATIONS; + const outerBreaks = this.breakStates; + const outerContinues = this.continueStates; + let converged = false; + this.quiet = true; + for (let iteration = 0; iteration < iterations; iteration++) { + const before = this.snapshot(); + const body = this.runBody(check); + this.restore(entry); + // The next iteration starts from any of: the head it started at, the end of a body that + // fell through, or a `continue`. A `break` is folded in as well, so that a range the + // analysis reports is never narrower than one the loop can really hold. + this.applyJoin([before, body.fellThrough, ...body.continued, ...body.broke]); + if (sameSnapshot(before, this.snapshot())) { + converged = true; + break; + } + } + if (!converged && trips > BigInt(MAX_LOOP_ITERATIONS)) { + this.checker.metricTable.widenedLoops++; + this.widenMutableIntegers(); + } + this.quiet = false; + + const head = this.snapshot(); + const final = this.runBody(check); + this.breakStates = outerBreaks; + this.continueStates = outerContinues; + this.restore(entry); + // After the loop the state is any of: never entered it (the head, which covers the entry), + // fell out of the last body, or left through a `break`. + this.applyJoin([head, final.fellThrough, ...final.broke]); + return final.statements; + } + + /** Runs a loop body once, keeping the states its `break`s and `continue`s left with. */ + private runBody(check: () => CStmt[]): { + statements: CStmt[]; + fellThrough: ScopeSnapshot; + broke: ScopeSnapshot[]; + continued: ScopeSnapshot[]; + } { + this.breakStates = []; + this.continueStates = []; + const statements = check(); + return { + statements, + fellThrough: this.snapshot(), + broke: this.breakStates, + continued: this.continueStates, + }; + } + + private widenMutableIntegers(): void { + for (const binding of this.scope.values()) { + if (binding.mutable && binding.type.kind === "Int") { + binding.type = tInt( + binding.type.lo < 0n ? SAFE_INT_LO : 0n, + binding.declared.kind === "Int" && binding.declared.hi < SAFE_INT_HI ? binding.declared.hi : SAFE_INT_HI, + ); + binding.declared = binding.type; + binding.widened = true; + } + if (binding.mutable && binding.type.kind === "List") { + binding.type = tList(binding.type.elem, 0, MAX_COLLECTION_LENGTH); + binding.declared = binding.type; + } + if (binding.mutable && binding.type.kind === "String") { + binding.type = tString(binding.type.cls, 0, MAX_COLLECTION_LENGTH); + binding.declared = binding.type; + } + } + } + + private noteWideInteger(type: SemType, span: Span): void { + if (type.kind === "Int" && (type.lo < SAFE_INT_LO || type.hi > SAFE_INT_HI) && !this.quiet) { + this.checker.metricTable.wideIntegers.push(`${span.file}: ${typeToString(type)}`); + } + } + + private requireBool(expr: CExpr, span: Span): void { + if (expr.type.kind !== "Bool") { + this.report( + "E_TRUTHINESS", + `only Bool is a condition, got ${typeToString(expr.type)}`, + span, + "compare explicitly; there is no truthiness in the subset", + ); + } + } + + /* ---------------------------------------------------------------- * + * Expressions + * ---------------------------------------------------------------- */ + + expr(node: HExpr, expected?: SemType): CExpr { + switch (node.kind) { + case "int": + return { kind: "lit", value: node.value, type: tInt(node.value, node.value), span: node.span }; + case "float": + return { kind: "lit", value: node.value, type: tFloat, span: node.span }; + case "bool": + return { kind: "lit", value: node.value, type: tBool, span: node.span }; + case "string": { + const target = expected?.kind === "Option" ? expected.inner : expected; + if (target?.kind === "Enum") { + if (!target.members.includes(node.value)) { + this.report( + "E_ENUM_MEMBER", + `${JSON.stringify(node.value)} is not one of ${target.members.map((member) => JSON.stringify(member)).join(" | ")}`, + node.span, + ); + } + return { + kind: "lit", + value: node.value, + type: tEnum(target.name, [node.value]), + span: node.span, + }; + } + return { kind: "lit", value: node.value, type: literalStringType(node.value), span: node.span }; + } + case "undefined": + return { kind: "none", type: tOption(expected?.kind === "Option" ? expected.inner : tNever), span: node.span }; + case "regex": + this.report( + "E_REGEX_VALUE", + "a regex is not a run-time value", + node.span, + "bind it to a module level constant and pass it to `re.test`", + ); + return { kind: "lit", value: node.source, type: tString("none"), span: node.span }; + case "name": + return this.name(node.name, node.span); + case "member": + return this.member(node); + case "index": + return this.index(node); + case "call": + return this.call(node); + case "binary": { + if (node.op === "===" || node.op === "!==") return this.equality(node); + const left = this.expr(node.left); + const right = this.expr(node.right, left.type.kind === "Float" ? tFloat : undefined); + return this.binaryOp(node.op, left, right, node.span); + } + case "logical": + return this.logical(node); + case "unary": { + if (node.op === "!") { + const operand = this.expr(node.operand, tBool); + this.requireBool(operand, node.span); + return { kind: "not", operand, type: tBool, span: node.span }; + } + const operand = this.expr(node.operand, expected); + return this.op(operand.type.kind === "Float" ? "float.neg" : "int.neg", [operand], node.span); + } + case "ternary": { + const test = this.withCondition(() => this.expr(node.test, tBool)); + this.requireBool(test, node.span); + const { whenTrue, whenFalse } = deriveNarrowing(test); + const entry = this.snapshot(); + this.applyNarrowing(whenTrue); + const then = this.expr(node.then, expected); + this.restore(entry); + this.applyNarrowing(whenFalse); + const otherwise = this.expr(node.otherwise, expected); + this.restore(entry); + return { + kind: "cond", + test, + then, + otherwise, + type: join(then.type, otherwise.type), + span: node.span, + }; + } + case "template": { + let result: CExpr | undefined; + for (const part of node.parts) { + const piece = + part.kind === "text" + ? ({ kind: "lit", value: part.value, type: literalStringType(part.value), span: node.span } as CExpr) + : this.expr(part.expr, tString("none")); + if (part.kind === "expr" && piece.type.kind !== "String" && piece.type.kind !== "Enum") { + this.report( + "E_INTERPOLATION", + `only strings interpolate, got ${typeToString(piece.type)}`, + node.span, + "convert explicitly, for example with `str.fromInt`", + ); + } + result = result === undefined ? piece : this.op("str.concat", [result, piece], node.span); + } + return result ?? { kind: "lit", value: "", type: tString("ascii", 0, 0), span: node.span }; + } + case "object": + return this.recordLiteral(node, expected); + case "array": { + const elemHint = expected?.kind === "List" ? expected.elem : undefined; + const items = node.items.map((item) => this.expr(item, elemHint)); + const elem = + elemHint ?? items.map((item) => item.type).reduce((left, right) => join(left, right), tNever); + return { + kind: "list", + items, + type: tList(elem, items.length, items.length), + span: node.span, + }; + } + case "lambda": + return this.lambda(node, expected); + case "new": + this.report( + "E_NEW", + "`new` is only allowed in `throw new SomeError(...)`", + node.span, + "records are built with object literals", + ); + return { kind: "lit", value: 0n, type: tNever, span: node.span }; + default: { + const exhaustive: never = node; + return exhaustive; + } + } + } + + private name(name: string, span: Span): CExpr { + const binding = this.scope.get(name); + if (binding !== undefined) { + this.noteRead(name, binding.type); + const local: CExpr = { kind: "local", name, type: binding.declared, span }; + if (binding.unwrapped) { + return { kind: "op", op: "opt.unwrap", args: [local], type: binding.type, span }; + } + return { kind: "local", name, type: binding.type, span }; + } + const qualified = this.checker.scopeOf(this.module).get(name); + const constant = qualified === undefined ? undefined : this.checker.constTable.get(qualified); + if (constant !== undefined) { + // Constants are comptime values, so they are baked into the Core where they are used. + return { kind: "lit", value: constant.value, type: constant.type, span }; + } + this.report("E_UNKNOWN_NAME", `unknown name \`${name}\``, span); + return { kind: "lit", value: 0n, type: tNever, span }; + } + + private member(node: Extract): CExpr { + if (node.target.kind === "name" && INTRINSIC_MODULES.has(node.target.name) && this.scope.get(node.target.name) === undefined) { + this.report( + "E_INTRINSIC_REF", + `\`${node.target.name}.${node.name}\` may only be called`, + node.span, + ); + return { kind: "lit", value: 0n, type: tNever, span: node.span }; + } + const target = this.expr(node.target); + if (node.name === "length") { + if (target.type.kind === "String") return this.op("str.len", [target], node.span); + if (target.type.kind === "List") return this.op("seq.len", [target], node.span); + } + if (target.type.kind === "Record") { + const record = this.checker.recordTable.get(target.type.name); + const field = record?.fields.find((item) => item.name === node.name); + if (field === undefined) { + this.report( + "E_UNKNOWN_FIELD", + `\`${target.type.name}\` has no field \`${node.name}\``, + node.span, + ); + return { kind: "lit", value: 0n, type: tNever, span: node.span }; + } + return { + kind: "field", + target, + name: node.name, + type: field.optional && field.type.kind !== "Option" ? tOption(field.type) : field.type, + span: node.span, + }; + } + if (target.type.kind === "Option") { + this.report( + "E_OPTION_ACCESS", + `\`${typeToString(target.type)}\` may be absent`, + node.span, + "narrow it first with `x === undefined`, or supply a default with `??`", + ); + return { kind: "lit", value: 0n, type: tNever, span: node.span }; + } + this.report("E_MEMBER", `${typeToString(target.type)} has no member \`${node.name}\``, node.span, METHOD_HELP[node.name]); + return { kind: "lit", value: 0n, type: tNever, span: node.span }; + } + + /** + * `forceChecked` is set by the `??` handling in `logical()`: `xs[i] ?? fallback` names the + * absent case itself, in the language's own terms, so it selects the checked accessor no + * matter what the checker can prove about `i` — provability only decides the accessor for a + * bare `xs[i]` used where a value (not an Option) is required, below. + */ + private index(node: Extract, forceChecked = false): CExpr { + const target = this.expr(node.target); + const index = this.expr(node.index, tIntDefault()); + if (target.type.kind === "String") { + if (!this.requireAsciiPositional(target, "`s[i]`", node.span)) { + return { kind: "lit", value: 0n, type: tNever, span: node.span }; + } + // JavaScript answers `undefined` past the end, which is exactly what `str.charAtOpt` + // answers; `str.charAt` is picked instead whenever the index is proven in range, so a + // caller that already has the proof gets the value back directly, not an Option of it. + return !forceChecked && provablyInRange(index, target.type) + ? this.op("str.charAt", [target, index], node.span) + : this.op("str.charAtOpt", [target, index], node.span); + } + if (target.type.kind === "List") { + // Same choice as above, between `seq.get` and its checked form `seq.at`: JavaScript + // answers `undefined` past the end where Go and Rust panic, so an unprovable index is a + // diagnostic-free Option here rather than a bug the compiler would otherwise have to + // reject outright. + return !forceChecked && provablyInRange(index, target.type) + ? this.op("seq.get", [target, index], node.span) + : this.op("seq.at", [target, index], node.span); + } + this.report("E_INDEX_TARGET", `${typeToString(target.type)} cannot be indexed`, node.span); + return { kind: "lit", value: 0n, type: tNever, span: node.span }; + } + + /** + * `charCodeAt`, `charAt`, `slice` and the bracket index all read a string by *position*, which + * only means the same thing in every target when the string is proven ASCII: JavaScript's + * position is a UTF-16 code unit, Python's is a code point and Go's is a byte, and the three + * disagree on every scalar above U+007F. An unproven string is a diagnostic here, never a + * silent, target-dependent lowering. + */ + private requireAsciiPositional(target: CExpr, construct: string, span: Span): boolean { + if (target.type.kind === "String" && target.type.cls !== "none") return true; + this.report( + "E_UTF16_POSITION", + `${construct} addresses a string by JavaScript's UTF-16 code unit; Python indexes the ` + + "same string by code point and Go by byte, and the three disagree above U+007F", + span, + "prove the string is ASCII first, for example a regex guard (`if (!PATTERN.test(value)) " + + "return …`) or `str.asAscii`, and only then does the position mean the same thing everywhere", + ); + return false; + } + + /** + * `toUpperCase`/`toLowerCase` only mean the same thing as `str.asciiUpper`/`str.asciiLower` + * once the string is proven ASCII (see `CASE_STRING_METHODS`); an unproven string is a + * diagnostic, never a silent lowering that would run different case tables in different + * targets. + */ + private requireAsciiCase(target: CExpr, construct: string, span: Span): boolean { + if (target.type.kind === "String" && target.type.cls !== "none") return true; + this.report( + "E_UNICODE_CASE", + `${construct} runs JavaScript's full Unicode case mapping, which touches scalars an ` + + "ASCII-only case fold leaves alone, and differs again by target outside U+007F", + span, + "prove the string is ASCII first, for example a regex guard (`if (!PATTERN.test(value)) " + + "return …`) or `str.asAscii`, and only then does casing mean the same thing everywhere", + ); + return false; + } + + private recordLiteral(node: Extract, expected?: SemType): CExpr { + const target = expected?.kind === "Option" ? expected.inner : expected; + if (target?.kind !== "Record") { + this.report( + "E_RECORD_TARGET", + "a record literal needs a declared record type here", + node.span, + "annotate the binding, the parameter or the return type", + ); + return { kind: "lit", value: 0n, type: tNever, span: node.span }; + } + const definition = this.checker.recordTable.get(target.name); + if (definition === undefined) { + this.report("E_UNKNOWN_TYPE", `unknown record \`${target.name}\``, node.span); + return { kind: "lit", value: 0n, type: tNever, span: node.span }; + } + const fields: { name: string; value: CExpr }[] = []; + for (const field of definition.fields) { + const provided = node.fields.find((item) => item.name === field.name); + const fieldType = field.optional && field.type.kind !== "Option" ? tOption(field.type) : field.type; + if (provided === undefined) { + if (!field.optional) { + this.report("E_MISSING_FIELD", `missing field \`${field.name}\``, node.span); + continue; + } + fields.push({ + name: field.name, + value: { kind: "none", type: fieldType, span: node.span }, + }); + continue; + } + const value = this.expr(provided.value, fieldType); + const coerced = + fieldType.kind === "Option" && value.type.kind !== "Option" + ? ({ kind: "some", inner: value, type: tOption(value.type), span: value.span } as CExpr) + : value; + if (!isSubtype(coerced.type, fieldType)) { + this.report( + "E_TYPE", + `field \`${field.name}\`: cannot assign ${typeToString(coerced.type)} to ${typeToString(fieldType)}`, + provided.span, + ); + } + fields.push({ name: field.name, value: coerced }); + } + for (const provided of node.fields) { + if (!definition.fields.some((field) => field.name === provided.name)) { + this.report( + "E_UNKNOWN_FIELD", + `\`${target.name}\` has no field \`${provided.name}\``, + provided.span, + ); + } + } + return { kind: "record", typeName: target.name, fields, type: target, span: node.span }; + } + + private lambda(node: Extract, expected?: SemType): CExpr { + const params = node.params.map((param, index) => { + const declared = + param.type === undefined ? undefined : this.checker.resolveType(param.type, param.name); + const hint = expected?.kind === "Lambda" ? expected.params[index] : undefined; + const type = declared ?? hint; + if (type === undefined) { + this.report("E_LAMBDA_PARAM", `cannot infer the type of \`${param.name}\``, param.span); + } + return { name: param.name, type: type ?? tNever }; + }); + + const inner = new Map(); + for (const [name, binding] of this.scope) { + // Closures capture values, never mutable slots: a captured local is immutable inside. + inner.set(name, { ...binding, mutable: false }); + } + for (const param of params) { + inner.set(param.name, { + name: param.name, + declared: param.type, + type: param.type, + mutable: false, + unwrapped: false, + }); + } + const expectedReturn = expected?.kind === "Lambda" ? expected.ret : tNever; + const sub = new FunctionChecker(this.checker, this.module, expectedReturn, inner); + sub.quiet = this.quiet; + // Shares the same observation map, not a copy: a closure that reads a captured bare-`number` + // parameter is still a use of it, and has to count toward the same inference the enclosing + // body's own reads do. + sub.paramObservations = this.paramObservations; + const body = sub.block(node.body); + this.addEffects(sub.effects); + const returnType = returnTypeOf(body); + return { + kind: "lambda", + params, + body, + type: tLambda( + params.map((param) => param.type), + returnType, + ), + span: node.span, + }; + } + + private logical(node: Extract): CExpr { + if (node.op === "??") { + // `xs[i] ?? fallback` / `s[i] ?? fallback`: under `noUncheckedIndexedAccess`, real + // TypeScript already types a bracket index `T | undefined` regardless of what the + // checker can prove about `i`, and writing `?? fallback` is the author's own statement + // that they want the absent case, not a claim about provability — so the bracket picks + // the checked accessor here unconditionally. A bare `xs[i]` elsewhere still uses + // provability (see `index()`), because there the author is asking for a value, not + // choosing between the two. + const left = node.left.kind === "index" ? this.index(node.left, true) : this.expr(node.left); + if (left.type.kind !== "Option") { + this.report( + "E_COALESCE", + `\`??\` applies to an Option, got ${typeToString(left.type)}`, + node.span, + ); + return left; + } + const right = this.expr(node.right, left.type.inner); + return this.op("opt.orElse", [left, right], node.span); + } + const left = this.expr(node.left, tBool); + this.requireBool(left, node.span); + const { whenTrue, whenFalse } = deriveNarrowing(left); + const entry = this.snapshot(); + this.applyNarrowing(node.op === "&&" ? whenTrue : whenFalse); + const right = this.expr(node.right, tBool); + this.requireBool(right, node.span); + this.restore(entry); + return node.op === "&&" + ? { kind: "and", left, right, type: tBool, span: node.span } + : { kind: "or", left, right, type: tBool, span: node.span }; + } + + private equality(node: Extract): CExpr { + const negated = node.op === "!=="; + const leftIsUndefined = node.left.kind === "undefined"; + const rightIsUndefined = node.right.kind === "undefined"; + if (leftIsUndefined || rightIsUndefined) { + const value = this.expr(leftIsUndefined ? node.right : node.left); + if (value.type.kind !== "Option") { + this.report( + "E_UNDEFINED_COMPARE", + `${typeToString(value.type)} is never undefined`, + node.span, + "only an Option is compared with undefined", + ); + } + const test = this.op("opt.isNone", [value], node.span); + return negated ? { kind: "not", operand: test, type: tBool, span: node.span } : test; + } + const left = this.expr(node.left); + const right = this.expr(node.right, left.type); + const test = this.op("core.eq", [left, right], node.span); + return negated ? { kind: "not", operand: test, type: tBool, span: node.span } : test; + } + + binaryOp(op: string, left: CExpr, right: CExpr, span: Span, negate = false): CExpr { + const kind = left.type.kind; + const table: Record = { + "+": "add", + "-": "sub", + "*": "mul", + "/": "div", + "%": "mod", + "<": "lt", + "<=": "le", + ">": "gt", + ">=": "ge", + }; + const name = table[op]; + if (name === undefined) { + this.report("E_OPERATOR", `unsupported operator \`${op}\``, span); + return left; + } + if (kind === "String") { + if (op === "+") return this.op("str.concat", [left, right], span); + this.report( + "E_STRING_OPERATOR", + `\`${op}\` on strings is outside the subset`, + span, + "use `str.compare`, whose order is the same in every target", + ); + return left; + } + if (kind === "Decimal") { + this.report( + "E_DECIMAL_OPERATOR", + `\`${op}\` on Decimal is outside the subset`, + span, + "use `dec.add`, `dec.sub`, `dec.mul` or `dec.divRound`, which name their scale and rounding", + ); + return left; + } + if (kind === "CivilDate") { + this.report( + "E_DATE_OPERATOR", + `\`${op}\` on CivilDate is outside the subset`, + span, + "use `date.compare`, `date.addDays` or `date.diffDays`", + ); + return left; + } + if (kind === "Float" && op === "%") { + this.report("E_FLOAT_MOD", "`%` on floats is outside the subset", span); + return left; + } + const prefix = kind === "Float" ? "float" : "int"; + const result = this.op(`${prefix}.${name}`, [left, right], span); + return negate ? { kind: "not", operand: result, type: tBool, span } : result; + } + + /** + * `Math.min`/`max`/`abs`/`trunc`/`floor`. Dispatched to `int.*`: every call site the source + * needs today is integer arithmetic, and there is no `float.*` counterpart yet — admitting one + * needs a second caller (docs/semantics.md, "Admission rule for intrinsics") — so a Float + * operand is rejected rather than silently handled by a made-up lowering. `trunc` and `floor` + * are the identity on an Int, which is already exact, so they never reach an intrinsic at all. + */ + private mathCall(method: string, args: readonly HExpr[], span: Span): CExpr { + const arity = method === "min" || method === "max" ? 2 : 1; + if (args.length !== arity) { + this.report("E_ARITY", `\`Math.${method}\` takes ${arity} argument(s), got ${args.length}`, span); + return { kind: "lit", value: 0n, type: tNever, span }; + } + const checked = args.map((arg) => this.expr(arg)); + const nonInt = checked.find((arg) => arg.type.kind !== "Int"); + if (nonInt !== undefined) { + this.report( + "E_MATH_FLOAT", + `\`Math.${method}\` on ${typeToString(nonInt.type)} is outside the subset: there is no \`float.${method}\``, + span, + "write the comparison explicitly, or keep the value an Int", + ); + return { kind: "lit", value: 0n, type: tNever, span }; + } + if (method === "trunc" || method === "floor") return checked[0]!; + return this.op(`int.${method}`, checked, span); + } + + /** + * `String(n)`: JavaScript's own stringification, admitted only for an Int argument, which is + * exactly `str.fromInt`. Anything else — a float, a boolean, an object — formats differently in + * every host and keeps the generic host-global rejection. + */ + private stringCall(args: readonly HExpr[], span: Span): CExpr { + if (args.length !== 1) { + this.report("E_ARITY", `\`String\` takes 1 argument, got ${args.length}`, span); + return { kind: "lit", value: "", type: tNever, span }; + } + const value = this.expr(args[0]!); + if (value.type.kind !== "Int") { + this.report( + "E_HOST_GLOBAL", + "`String(...)` is JavaScript's own stringification, which formats a float, a boolean " + + "or a record each in its own host-specific way", + span, + "convert explicitly: `str.fromInt` for an Int, or write the formatting in source for anything else", + ); + return { kind: "lit", value: "", type: tNever, span }; + } + return this.op("str.fromInt", [value], span); + } + + /** + * `value.replace(/[^0-9]/g, "")`: JavaScript's `.replace` runs its own replacement algorithm + * (capture group substitution, a callback, only the first match without `/g/`), which nothing + * else has to reproduce identically. The one shape that is target-independent is dropping every + * scalar outside a class — a global match of a single negated class against an empty + * replacement — which is exactly `re.retain` on the class's own (un-negated) content. Anything + * else about the call falls through to `undefined`, which the caller routes to the ordinary + * method-sugar rejection. + */ + private retainFromReplace(target: HExpr, args: readonly HExpr[], span: Span): CExpr | undefined { + if (args.length !== 2) return undefined; + const pattern = args[0]!; + const replacement = args[1]!; + if (pattern.kind !== "regex" || replacement.kind !== "string" || replacement.value !== "") return undefined; + if (pattern.flags !== "g") { + this.report( + "E_REPLACE_UNSUPPORTED", + "`.replace` without the `g` flag rewrites only the first match, which the engine has " + + "no target-independent way to reproduce", + span, + "add the `g` flag if you meant to remove every occurrence, which is `re.retain` on the class", + ); + return { kind: "lit", value: "", type: tNever, span }; + } + const inner = negatedClassInner(pattern.source); + if (inner === undefined) { + this.report( + "E_REPLACE_UNSUPPORTED", + "`.replace` runs JavaScript's own replacement algorithm, which has no shared meaning " + + 'across targets; the one shape the engine accepts is `value.replace(/[^…]/g, "")`, ' + + "removing every scalar outside a single class", + span, + "write the pattern as one negated character class, or perform the substitution in source", + ); + return { kind: "lit", value: "", type: tNever, span }; + } + const retained: HExpr = { kind: "regex", source: `^[${inner}]$`, flags: "", span: pattern.span }; + return this.intrinsicCall("re.retain", [retained, target], span); + } + + private call(node: Extract): CExpr { + const callee = node.callee; + + // Intrinsic module call: `str.codeAt(value, index)`. + if ( + callee.kind === "member" && + callee.target.kind === "name" && + INTRINSIC_MODULES.has(callee.target.name) && + this.scope.get(callee.target.name) === undefined + ) { + return this.intrinsicCall(`${callee.target.name}.${callee.name}`, node.args, node.span); + } + + // `Math.min`/`max`/`abs`/`trunc`/`floor`, rewritten by the frontend into this same shape. + if ( + callee.kind === "member" && + callee.target.kind === "name" && + callee.target.name === "Math" && + this.scope.get("Math") === undefined + ) { + return this.mathCall(callee.name, node.args, node.span); + } + + // `String(n)`: sound only for an Int argument, where it is exactly `str.fromInt`. + if (callee.kind === "name" && callee.name === "String" && this.scope.get("String") === undefined) { + return this.stringCall(node.args, node.span); + } + + // `value.replace(/[^0-9]/g, "")`: the one `.replace` shape the engine can prove sound, + // which is `re.retain` on the class's own content. Any other `.replace` call falls through + // to the method sugar below, which rejects it. + if (callee.kind === "member" && callee.name === "replace") { + const retained = this.retainFromReplace(callee.target, node.args, node.span); + if (retained !== undefined) return retained; + } + + // `PATTERN.test(value)`, where `PATTERN` is a regex literal or a module level constant + // bound to one: the same test `re.test(PATTERN, value)` performs. + if (callee.kind === "member" && callee.name === "test" && this.regexOf(callee.target) !== undefined) { + return this.intrinsicCall("re.test", [callee.target, ...node.args], node.span); + } + + // Method sugar on a value: `value.slice(0, 9)`, `list.map(f)`. + if (callee.kind === "member") { + return this.methodCall(callee, node.args, node.span); + } + + if (callee.kind !== "name") { + this.report("E_CALLEE", "only named functions and intrinsics may be called", node.span); + return { kind: "lit", value: 0n, type: tNever, span: node.span }; + } + + const qualified = this.checker.scopeOf(this.module).get(callee.name); + const signature = qualified === undefined ? undefined : this.checker.signatureOf(qualified); + if (signature === undefined || qualified === undefined) { + this.report("E_UNKNOWN_FUNCTION", `unknown function \`${callee.name}\``, node.span); + return { kind: "lit", value: 0n, type: tNever, span: node.span }; + } + if (node.args.length !== signature.params.length) { + this.report( + "E_ARITY", + `\`${callee.name}\` takes ${signature.params.length} argument(s), got ${node.args.length}`, + node.span, + ); + } + const args = signature.params.map((param, index) => { + const argument = node.args[index]; + if (argument === undefined) return { kind: "lit", value: 0n, type: tNever, span: node.span } as CExpr; + const checked = this.expr(argument, param.type); + const coerced = + param.type.kind === "Option" && checked.type.kind !== "Option" && checked.type.kind !== "Never" + ? ({ kind: "some", inner: checked, type: tOption(checked.type), span: checked.span } as CExpr) + : checked; + if (!isSubtype(coerced.type, param.type) && !fitsPlatformDomain(coerced.type, param.type)) { + this.report( + "E_ARGUMENT", + `\`${callee.name}\`: argument \`${param.name}\` expects ${typeToString(param.type)}, got ${typeToString(coerced.type)}`, + argument.span, + ); + } + return coerced; + }); + const instance = this.checker.instantiate(qualified, args.map((arg) => arg.type)); + if (instance === undefined) { + this.report("E_UNKNOWN_FUNCTION", `unknown function \`${callee.name}\``, node.span); + return { kind: "lit", value: 0n, type: tNever, span: node.span }; + } + this.addEffects(instance.effects); + return { kind: "call", fn: instance.name, args, type: instance.ret, span: node.span }; + } + + private methodCall( + callee: Extract, + args: readonly HExpr[], + span: Span, + ): CExpr { + const target = this.expr(callee.target); + const method = callee.name; + if (target.type.kind === "Int" && method === "toString") { + if (args.length !== 0) { + this.report( + "E_METHOD", + "`toString` with a radix is outside the subset: every target formats an Int in base 10", + span, + "use `str.fromInt`, which is always base 10", + ); + return { kind: "lit", value: "", type: tNever, span }; + } + return this.op("str.fromInt", [target], span); + } + if (target.type.kind === "String") { + const intrinsic = STRING_METHODS[method]; + if (intrinsic === undefined) { + this.report("E_METHOD", `\`${method}\` is outside the subset`, span, METHOD_HELP[method]); + return { kind: "lit", value: 0n, type: tNever, span }; + } + if (POSITIONAL_STRING_METHODS.has(method) && !this.requireAsciiPositional(target, `\`${method}\``, span)) { + return { kind: "lit", value: 0n, type: tNever, span }; + } + if (CASE_STRING_METHODS.has(method) && !this.requireAsciiCase(target, `\`${method}\``, span)) { + return { kind: "lit", value: 0n, type: tNever, span }; + } + return this.intrinsicCallWith( + intrinsic, + [target, ...args.map((arg) => (expected?: SemType) => this.expr(arg, expected))], + span, + ); + } + if (target.type.kind === "List") { + if (method === "reduce") { + // `xs.reduce(f, init)` is `seq.fold(xs, init, f)`: the Core names the initial value first. + const initial = this.expr(args[1]!); + const fn = this.expr(args[0]!, tLambda([initial.type, target.type.elem], initial.type)); + return this.op("seq.fold", [target, initial, fn], span); + } + const intrinsic = LIST_METHODS[method]; + if (intrinsic === undefined) { + this.report("E_METHOD", `\`${method}\` is outside the subset`, span, METHOD_HELP[method]); + return { kind: "lit", value: 0n, type: tNever, span }; + } + return this.intrinsicCallWith( + intrinsic, + [target, ...args.map((arg) => (expected?: SemType) => this.expr(arg, expected))], + span, + ); + } + this.report("E_METHOD", `${typeToString(target.type)} has no method \`${method}\``, span, METHOD_HELP[method]); + return { kind: "lit", value: 0n, type: tNever, span }; + } + + private intrinsicCall(name: string, args: readonly HExpr[], span: Span): CExpr { + if (name === "re.retain") { + const pattern = args[0]; + const regex = pattern === undefined ? undefined : this.regexOf(pattern); + const singleClass = regex === undefined ? undefined : singleClassOf(regex); + if (regex === undefined || singleClass === undefined) { + this.report( + "E_RETAIN_PATTERN", + "`re.retain` needs a constant bound to a regex that is exactly one character class", + span, + "declare `const DIGIT = /^[0-9]$/;` and pass it", + ); + return { kind: "lit", value: "", type: tNever, span }; + } + const subject = this.expr(args[1]!); + const result = this.op("re.retain", [subject], span); + const subjectLength = subject.type.kind === "String" ? subject.type.max : MAX_COLLECTION_LENGTH; + return { + ...result, + regex, + type: tString(singleClass, 0, subjectLength), + } as CExpr; + } + if (name === "re.test") { + const pattern = args[0]; + const regex = pattern === undefined ? undefined : this.regexOf(pattern); + if (regex === undefined) { + this.report( + "E_REGEX_ARGUMENT", + "`re.test` needs a regex literal or a constant bound to one", + span, + "declare `const PATTERN = /…/;` at module level", + ); + return { kind: "lit", value: false, type: tBool, span }; + } + const subject = this.expr(args[1]!); + const result = this.op("re.test", [subject], span); + return { ...result, regex } as CExpr; + } + return this.intrinsicCallWith( + name, + args.map((arg) => (expected?: SemType) => this.expr(arg, expected)), + span, + ); + } + + /** + * Checks an intrinsic call left to right, so a combinator can hand the element type to the + * lambda that follows it. + */ + private intrinsicCallWith( + name: string, + args: readonly (CExpr | ((expected?: SemType) => CExpr))[], + span: Span, + ): CExpr { + const definition = lookupIntrinsic(name); + if (definition === undefined) { + this.report("E_UNKNOWN_INTRINSIC", `unknown intrinsic \`${name}\``, span); + return { kind: "lit", value: 0n, type: tNever, span }; + } + const checked: CExpr[] = []; + for (const [index, argument] of args.entries()) { + if (typeof argument !== "function") { + checked.push(argument); + continue; + } + const prior = checked.map((item) => item.type); + const lambdaHint = + definition.lambdaParams === undefined + ? undefined + : safely(() => definition.lambdaParams!(prior, index)); + const hint = + lambdaHint === undefined + ? definition.paramHint?.(index, prior) + : tLambda(lambdaHint, tNever); + checked.push(argument(hint)); + } + if (name === "task.race") { + this.checkRaceTasks(checked, span); + } + this.addEffects(definition.effects); + return this.op(name, checked, span); + } + + /** Builds an op node, turning a signature failure into a span-aware diagnostic. */ + op(name: string, args: readonly CExpr[], span: Span): CExpr { + const definition = lookupIntrinsic(name); + if (definition === undefined) { + this.report("E_UNKNOWN_INTRINSIC", `unknown intrinsic \`${name}\``, span); + return { kind: "lit", value: 0n, type: tNever, span }; + } + try { + const type = definition.signature(args.map((arg) => arg.type)); + this.addEffects(definition.effects); + this.noteWideInteger(type, span); + return { kind: "op", op: name, args, type, span }; + } catch (error) { + if (error instanceof SignatureError) { + this.report("E_SIGNATURE", error.message, span, error.suggestion); + return { kind: "lit", value: 0n, type: tNever, span }; + } + throw error; + } + } + + /** + * A race discards the losers, so only idempotent work belongs inside one: a task may read + * (an Http GET) and may wait, but it may not consume randomness or send anything. + */ + private checkRaceTasks(args: readonly CExpr[], span: Span): void { + const list = args[0]; + if (list === undefined || list.kind !== "list") return; + for (const task of list.items) { + if (task.kind !== "lambda") continue; + // A task usually delegates to a function, so the walk follows calls: the rule is about + // what the task *does*, not about where it is written. + for (const op of this.reachableOperations(task.body)) { + if (op.op === "random.nextU32") { + this.report( + "E_RACE_EFFECT", + "a task inside `task.race` may not use Random: a losing task's draw would be discarded", + op.span, + "draw before the race and pass the value in", + ); + } + if (op.op !== "http.request") continue; + const request = op.args[0]; + const method = + request?.kind === "record" + ? request.fields.find((field) => field.name === "method")?.value + : undefined; + if (method?.kind === "lit" && method.value !== "GET") { + this.report( + "E_RACE_EFFECT", + `a task inside \`task.race\` may only perform an idempotent request, not ${String(method.value)}`, + op.span, + "race the reads, and perform the write once the winner is known", + ); + } + } + } + } + + /** Every operation a body performs, following calls into the functions it reaches. */ + private reachableOperations(body: readonly CStmt[]): Extract[] { + const seen = new Set(); + const collect = (statements: readonly CStmt[]): Extract[] => { + const operations = operationsOf(statements); + for (const callee of callsOf(statements)) { + if (seen.has(callee)) continue; + seen.add(callee); + const calleeBody = this.checker.bodyOf(callee); + if (calleeBody !== undefined) operations.push(...collect(calleeBody)); + } + return operations; + }; + return collect(body); + } + + private regexOf(node: HExpr): NormalizedRegex | undefined { + if (node.kind === "regex") { + try { + return normalizeRegex(node.source, node.span); + } catch (error) { + this.report("E_REGEX", error instanceof RegexError ? error.message : String(error), node.span); + return undefined; + } + } + if (node.kind === "name") { + const qualified = this.checker.scopeOf(this.module).get(node.name); + const constant = qualified === undefined ? undefined : this.checker.constTable.get(qualified); + return constant?.regex; + } + return undefined; + } +} + +/* ==================================================================== * + * Helpers + * ==================================================================== */ + +function safely(compute: () => T): T | undefined { + try { + return compute(); + } catch { + return undefined; + } +} + +function literalStringType(value: string): SemType { + const points = [...value].map((scalar) => scalar.codePointAt(0)!); + const cls = + points.length > 0 && points.every((point) => point >= 0x30 && point <= 0x39) + ? "digits" + : points.every((point) => point < 0x80) + ? "ascii" + : "none"; + return tString(cls, points.length, points.length); +} + +function returnTypeOf(body: readonly CStmt[]): SemType { + let result: SemType = tNever; + const visit = (statements: readonly CStmt[]): void => { + for (const statement of statements) { + switch (statement.kind) { + case "return": + result = join(result, statement.value?.type ?? tVoid); + break; + case "if": + visit(statement.then); + visit(statement.otherwise); + break; + case "switch": + for (const entry of statement.cases) visit(entry.body); + if (statement.otherwise !== undefined) visit(statement.otherwise); + break; + case "forRange": + case "forEach": + visit(statement.body); + break; + default: + break; + } + } + }; + visit(body); + return result; +} + +function exitsScope(body: readonly CStmt[]): boolean { + return body.some( + (statement) => + statement.kind === "return" || + statement.kind === "fail" || + statement.kind === "break" || + statement.kind === "continue", + ); +} + +/** + * Whether a value fits a declared integer type once the platform domain is taken into account. + * + * A value that walks a collection can exceed the platform domain only by a bounded step, because + * every loop's trip count is bounded by MAX_COLLECTION_LENGTH; the same reasoning that lets a + * widened loop counter be clamped lets such a value be passed and returned. + */ +export function fitsPlatformDomain(actual: SemType, declared: SemType): boolean { + if (actual.kind !== "Int" || declared.kind !== "Int") return false; + if (declared.hi < SAFE_INT_HI - MAX_WIDENED_STEP) return false; + const step = stepSize(declared, actual); + return step !== undefined && step <= MAX_WIDENED_STEP; +} + +/** Whether `type` is exactly the unconstrained platform-safe domain: an `Int` no guard narrowed. */ +function isUnconstrainedInt(type: SemType): boolean { + return type.kind === "Int" && type.lo === SAFE_INT_LO && type.hi === SAFE_INT_HI; +} + +/** How far one assignment moves a binding's range, when both are integers. */ +function stepSize(declared: SemType, assigned: SemType): bigint | undefined { + if (declared.kind !== "Int" || assigned.kind !== "Int") return undefined; + const low = declared.lo - assigned.lo; + const high = assigned.hi - declared.hi; + const step = (low > high ? low : high); + return step < 0n ? 0n : step; +} + +function widenAssignment(declared: SemType): SemType { + // An accumulator keeps its declared shape; only its refinements move, and the loop fixpoint + // is what proves where they land. + return declared; +} + +function sameSnapshot(left: ScopeSnapshot, right: ScopeSnapshot): boolean { + if (left.size !== right.size) return false; + for (const [name, state] of left) { + const other = right.get(name); + if (other === undefined) return false; + if (typeToString(state.type) !== typeToString(other.type)) return false; + if (state.unwrapped !== other.unwrapped) return false; + } + return true; +} + +/* ---------------------------------------------------------------- * + * Flow-sensitive narrowing + * ---------------------------------------------------------------- */ + +function mergeNarrowings(left: Narrowing, right: Narrowing): Narrowing { + const merged: Narrowing = new Map(left); + for (const [name, type] of right) { + const existing = merged.get(name); + merged.set(name, existing === undefined ? type : intersectTypes(existing, type)); + } + return merged; +} + +/** Keeps only the facts both sides establish, widened to cover either. */ +function hullNarrowings(left: Narrowing, right: Narrowing): Narrowing { + const merged: Narrowing = new Map(); + for (const [name, type] of left) { + const other = right.get(name); + if (other !== undefined) merged.set(name, join(type, other)); + } + return merged; +} + +function intersectTypes(left: SemType, right: SemType): SemType { + if (left.kind === "String" && right.kind === "String") { + return tString( + left.cls === "digits" || right.cls === "digits" ? "digits" : left.cls === "ascii" || right.cls === "ascii" ? "ascii" : "none", + Math.max(left.min, right.min), + Math.min(left.max, right.max), + left.pattern ?? right.pattern, + ); + } + if (left.kind === "Int" && right.kind === "Int") { + return tInt(left.lo > right.lo ? left.lo : right.lo, left.hi < right.hi ? left.hi : right.hi); + } + if (left.kind === "List" && right.kind === "List") { + return tList(left.elem, Math.max(left.min, right.min), Math.min(left.max, right.max)); + } + return right; +} + +/** The facts a condition establishes on each side of a branch. */ +export function deriveNarrowing(test: CExpr): { whenTrue: Narrowing; whenFalse: Narrowing } { + const empty = { whenTrue: new Map(), whenFalse: new Map() } as { + whenTrue: Narrowing; + whenFalse: Narrowing; + }; + switch (test.kind) { + case "not": { + const inner = deriveNarrowing(test.operand); + return { whenTrue: inner.whenFalse, whenFalse: inner.whenTrue }; + } + case "and": { + const left = deriveNarrowing(test.left); + const right = deriveNarrowing(test.right); + return { whenTrue: mergeNarrowings(left.whenTrue, right.whenTrue), whenFalse: new Map() }; + } + case "or": { + const left = deriveNarrowing(test.left); + const right = deriveNarrowing(test.right); + // When either side may be the one that held, the fact is the hull of the two: an + // interval domain can say "one of these two ranges" only as the range that covers both. + return { + whenTrue: hullNarrowings(left.whenTrue, right.whenTrue), + whenFalse: mergeNarrowings(left.whenFalse, right.whenFalse), + }; + } + case "op": + return narrowFromOp(test) ?? empty; + default: + return empty; + } +} + +/** + * The local an expression names, looking through the unwrap the checker inserts once an Option + * has been proven present. Narrowing a value does not stop being possible because it came out of + * an Option. + */ +function localOf(expr: CExpr): Extract | undefined { + if (expr.kind === "local") return expr; + if (expr.kind === "op" && expr.op === "opt.unwrap") return localOf(expr.args[0]!); + return undefined; +} + +function narrowFromOp(test: Extract): { whenTrue: Narrowing; whenFalse: Narrowing } | undefined { + const whenTrue: Narrowing = new Map(); + const whenFalse: Narrowing = new Map(); + + if (test.op === "re.test" && test.regex !== undefined) { + const subject = localOf(test.args[0]!); + if (subject === undefined) return undefined; + const regex = test.regex; + whenTrue.set( + subject.name, + tString( + regex.digitsOnly ? "digits" : regex.asciiOnly ? "ascii" : "none", + regex.minLength, + regex.maxLength, + regex.source, + ), + ); + return { whenTrue, whenFalse }; + } + + if (test.op === "opt.isNone") { + const subject = test.args[0]!; + if (subject.kind === "local" && subject.type.kind === "Option") { + whenFalse.set(subject.name, subject.type.inner); + } + return { whenTrue, whenFalse }; + } + + if (test.op === "core.eq") { + const [left, right] = test.args; + if (left === undefined || right === undefined) return undefined; + // `value === "v2"` narrows an enum to the matched member. + if (left.kind === "local" && left.type.kind === "Enum" && right.kind === "lit") { + whenTrue.set(left.name, tEnum(left.type.name, [String(right.value)])); + if (left.type.members.length === 2) { + whenFalse.set( + left.name, + tEnum( + left.type.name, + left.type.members.filter((member) => member !== String(right.value)), + ), + ); + } + return { whenTrue, whenFalse }; + } + return narrowLength(left, right, "eq", whenTrue, whenFalse); + } + + if (["int.lt", "int.le", "int.gt", "int.ge"].includes(test.op)) { + const [left, right] = test.args; + if (left === undefined || right === undefined) return undefined; + return narrowLength(left, right, test.op.slice(4) as "lt" | "le" | "gt" | "ge", whenTrue, whenFalse); + } + + return undefined; +} + +/** Narrows either an integer local or the length of the string/list a `length` call names. */ +function narrowLength( + left: CExpr, + right: CExpr, + comparison: "eq" | "lt" | "le" | "gt" | "ge", + whenTrue: Narrowing, + whenFalse: Narrowing, +): { whenTrue: Narrowing; whenFalse: Narrowing } | undefined { + if (right.type.kind !== "Int" || right.type.lo !== right.type.hi) return { whenTrue, whenFalse }; + const bound = right.type.lo; + + /** + * `x !== k` only narrows when `k` sits at an end of the current range: the remainder is then + * still an interval, which is what an interval domain can represent. + */ + const excludeEndpoint = (lo: bigint, hi: bigint): { lo: bigint; hi: bigint } => ({ + lo: lo === bound ? bound + 1n : lo, + hi: hi === bound ? bound - 1n : hi, + }); + + const trueRange = (): { lo: bigint; hi: bigint } => { + switch (comparison) { + case "eq": + return { lo: bound, hi: bound }; + case "lt": + return { lo: -(10n ** 30n), hi: bound - 1n }; + case "le": + return { lo: -(10n ** 30n), hi: bound }; + case "gt": + return { lo: bound + 1n, hi: 10n ** 30n }; + case "ge": + return { lo: bound, hi: 10n ** 30n }; + default: { + const exhaustive: never = comparison; + return exhaustive; + } + } + }; + const falseRange = (): { lo: bigint; hi: bigint } => { + switch (comparison) { + case "eq": + return { lo: -(10n ** 30n), hi: 10n ** 30n }; + case "lt": + return { lo: bound, hi: 10n ** 30n }; + case "le": + return { lo: bound + 1n, hi: 10n ** 30n }; + case "gt": + return { lo: -(10n ** 30n), hi: bound }; + case "ge": + return { lo: -(10n ** 30n), hi: bound - 1n }; + default: { + const exhaustive: never = comparison; + return exhaustive; + } + } + }; + + const leftLocal = localOf(left); + if (leftLocal !== undefined && left.type.kind === "Int") { + const current = left.type; + const positive = trueRange(); + const negative = comparison === "eq" ? excludeEndpoint(current.lo, current.hi) : falseRange(); + whenTrue.set( + leftLocal.name, + tInt( + current.lo > positive.lo ? current.lo : positive.lo, + current.hi < positive.hi ? current.hi : positive.hi, + ), + ); + whenFalse.set( + leftLocal.name, + tInt( + current.lo > negative.lo ? current.lo : negative.lo, + current.hi < negative.hi ? current.hi : negative.hi, + ), + ); + return { whenTrue, whenFalse }; + } + + if (left.kind === "op" && (left.op === "str.len" || left.op === "seq.len")) { + const subject = localOf(left.args[0]!); + if (subject === undefined) return { whenTrue, whenFalse }; + const positive = trueRange(); + const subjectLengths = left.args[0]!.type; + const lengths = + subjectLengths.kind === "String" || subjectLengths.kind === "List" + ? { lo: BigInt(subjectLengths.min), hi: BigInt(subjectLengths.max) } + : { lo: 0n, hi: 0n }; + const negative = comparison === "eq" ? excludeEndpoint(lengths.lo, lengths.hi) : falseRange(); + const clampLow = (value: bigint): number => Number(value < 0n ? 0n : value); + const clampHigh = (value: bigint): number => + Number(value > BigInt(MAX_COLLECTION_LENGTH) ? BigInt(MAX_COLLECTION_LENGTH) : value); + const subjectType = left.args[0]!.type; + if (subjectType.kind === "String") { + const current = subjectType; + whenTrue.set( + subject.name, + tString( + current.cls, + Math.max(current.min, clampLow(positive.lo)), + Math.min(current.max, clampHigh(positive.hi)), + current.pattern, + ), + ); + whenFalse.set( + subject.name, + tString( + current.cls, + Math.max(current.min, clampLow(negative.lo)), + Math.min(current.max, clampHigh(negative.hi)), + current.pattern, + ), + ); + } else if (subjectType.kind === "List") { + const current = subjectType; + whenTrue.set( + subject.name, + tList(current.elem, Math.max(current.min, clampLow(positive.lo)), Math.min(current.max, clampHigh(positive.hi))), + ); + whenFalse.set( + subject.name, + tList(current.elem, Math.max(current.min, clampLow(negative.lo)), Math.min(current.max, clampHigh(negative.hi))), + ); + } + return { whenTrue, whenFalse }; + } + + return { whenTrue, whenFalse }; +} diff --git a/engine/src/core/ir.ts b/engine/src/core/ir.ts new file mode 100644 index 000000000..4193f38a7 --- /dev/null +++ b/engine/src/core/ir.ts @@ -0,0 +1,361 @@ +/** + * Core IR: what the program means. + * + * Structured and typed, with no SSA, no CFG and nothing machine-like. Structured control flow is + * preserved on purpose: it is the information a source-level backend needs to print readable + * native code. Every node carries its semantic type; every function carries its effect set. + */ + +import type { Span } from "../diagnostics.ts"; +import type { EffectSet } from "../effects.ts"; +import type { NormalizedRegex } from "../regex.ts"; +import type { SemType } from "../types.ts"; +import { typeToString } from "../types.ts"; +import type { Value } from "../values.ts"; + +export type CRecordDef = { + readonly name: string; + readonly fields: readonly { + readonly name: string; + readonly type: SemType; + readonly optional: boolean; + readonly doc?: string; + }[]; + readonly doc?: string; + readonly exported: boolean; +}; + +export type CErrorDef = { + readonly name: string; + /** The nearest declared ancestor, or `undefined` for a root domain error. */ + readonly base?: string; + readonly doc?: string; + readonly exported: boolean; +}; + +export type CExpr = + | { readonly kind: "lit"; readonly value: Value; readonly type: SemType; readonly span: Span } + | { readonly kind: "local"; readonly name: string; readonly type: SemType; readonly span: Span } + | { readonly kind: "none"; readonly type: SemType; readonly span: Span } + | { readonly kind: "some"; readonly inner: CExpr; readonly type: SemType; readonly span: Span } + | { + readonly kind: "record"; + readonly typeName: string; + readonly fields: readonly { readonly name: string; readonly value: CExpr }[]; + readonly type: SemType; + readonly span: Span; + } + | { + readonly kind: "field"; + readonly target: CExpr; + readonly name: string; + readonly type: SemType; + readonly span: Span; + } + | { readonly kind: "list"; readonly items: readonly CExpr[]; readonly type: SemType; readonly span: Span } + | { + readonly kind: "call"; + readonly fn: string; + readonly args: readonly CExpr[]; + readonly type: SemType; + readonly span: Span; + } + | { + readonly kind: "op"; + readonly op: string; + readonly args: readonly CExpr[]; + /** A comptime payload, currently only the normalized regex of `re.test`. */ + readonly regex?: NormalizedRegex; + readonly type: SemType; + readonly span: Span; + } + | { + readonly kind: "lambda"; + readonly params: readonly { readonly name: string; readonly type: SemType }[]; + readonly body: readonly CStmt[]; + readonly type: SemType; + readonly span: Span; + } + | { + readonly kind: "cond"; + readonly test: CExpr; + readonly then: CExpr; + readonly otherwise: CExpr; + readonly type: SemType; + readonly span: Span; + } + | { + readonly kind: "and"; + readonly left: CExpr; + readonly right: CExpr; + readonly type: SemType; + readonly span: Span; + } + | { + readonly kind: "or"; + readonly left: CExpr; + readonly right: CExpr; + readonly type: SemType; + readonly span: Span; + } + | { readonly kind: "not"; readonly operand: CExpr; readonly type: SemType; readonly span: Span }; + +export type CStmt = + | { + readonly kind: "let"; + readonly name: string; + readonly mutable: boolean; + readonly init: CExpr; + readonly type: SemType; + readonly span: Span; + } + | { readonly kind: "assign"; readonly name: string; readonly value: CExpr; readonly span: Span } + | { + readonly kind: "setIndex"; + readonly name: string; + readonly index: CExpr; + readonly value: CExpr; + readonly span: Span; + } + | { readonly kind: "push"; readonly name: string; readonly value: CExpr; readonly span: Span } + | { + readonly kind: "if"; + readonly test: CExpr; + readonly then: readonly CStmt[]; + readonly otherwise: readonly CStmt[]; + readonly span: Span; + } + | { + readonly kind: "switch"; + readonly subject: CExpr; + readonly cases: readonly { readonly values: readonly Value[]; readonly body: readonly CStmt[] }[]; + readonly otherwise?: readonly CStmt[]; + readonly span: Span; + } + | { + readonly kind: "forRange"; + readonly name: string; + readonly type: SemType; + readonly from: CExpr; + readonly to: CExpr; + readonly inclusive: boolean; + readonly step: bigint; + readonly body: readonly CStmt[]; + readonly span: Span; + } + | { + readonly kind: "forEach"; + readonly name: string; + readonly type: SemType; + readonly iterable: CExpr; + readonly body: readonly CStmt[]; + readonly span: Span; + } + | { readonly kind: "return"; readonly value?: CExpr; readonly span: Span } + | { + readonly kind: "fail"; + readonly errorClass: string; + readonly args: readonly CExpr[]; + readonly span: Span; + } + | { readonly kind: "break"; readonly span: Span } + | { readonly kind: "continue"; readonly span: Span } + | { readonly kind: "expr"; readonly expr: CExpr; readonly span: Span }; + +export type CParam = { readonly name: string; readonly type: SemType; readonly doc?: string }; + +export type CFunc = { + /** Fully qualified: `::`. */ + readonly name: string; + readonly module: string; + readonly localName: string; + readonly params: readonly CParam[]; + readonly ret: SemType; + readonly effects: EffectSet; + readonly body: readonly CStmt[]; + /** + * True for a utility: an exported function of a source-root module, the published core API. + * Drives what a backend treats as a driver entry point and lists in `API.json`; unrelated to + * whether the function's own source module exported it (see `moduleExported`) — a library + * helper can be reachable across modules without ever being a utility. + */ + readonly exported: boolean; + /** + * True when the source module that declares this function marked it `export`, whatever module + * that is — root or `lib/`. This is the literal per-file fact a backend's printer turns into + * `export`/`pub`/no leading underscore, so a generated module's public surface matches its + * source module's, name for name. False for a specialization (checked against a call site's + * argument types, so no single generated name is "the" export) even when its declaration was. + */ + readonly moduleExported: boolean; + readonly doc?: string; + readonly span: Span; + /** Functions this one calls, fully qualified. */ + readonly calls: readonly string[]; + /** True once capability threading has decided this function receives the environment. */ + readonly usesEnv: boolean; +}; + +export type CConst = { + readonly name: string; + readonly type: SemType; + readonly value: Value; + readonly module: string; + /** Present when the constant is a regex literal, which is comptime-only data. */ + readonly regex?: NormalizedRegex; +}; + +export type CProgram = { + readonly records: ReadonlyMap; + readonly errors: ReadonlyMap; + readonly functions: ReadonlyMap; + readonly consts: ReadonlyMap; + /** Exported utilities, in declaration order: the roots of every dependency closure. */ + readonly entryPoints: readonly string[]; +}; + +export function mapStatements( + body: readonly CStmt[], + visit: (statement: CStmt) => CStmt[], +): CStmt[] { + return body.flatMap((statement) => visit(statement)); +} + +/** A readable, reviewable dump of the annotated Core. This is what `--dump core` prints. */ +export function dumpProgram(program: CProgram): string { + const lines: string[] = []; + for (const record of program.records.values()) { + lines.push( + `record ${record.name} { ${record.fields.map((field) => `${field.name}${field.optional ? "?" : ""}: ${typeToString(field.type)}`).join(", ")} }`, + ); + } + for (const error of program.errors.values()) { + lines.push(`error ${error.name}${error.base === undefined ? "" : ` extends ${error.base}`}`); + } + for (const constant of program.consts.values()) { + lines.push(`const ${constant.name}: ${typeToString(constant.type)}`); + } + for (const fn of program.functions.values()) { + lines.push(""); + lines.push( + `fn ${fn.name}(${fn.params.map((param) => `${param.name}: ${typeToString(param.type)}`).join(", ")}): ${typeToString(fn.ret)} ! ${effectLabel(fn)}`, + ); + for (const statement of fn.body) lines.push(...dumpStmt(statement, 1)); + } + return lines.join("\n"); +} + +function effectLabel(fn: CFunc): string { + const parts: string[] = []; + for (const name of fn.effects.fail) parts.push(`Fail<${name}>`); + if (fn.effects.http) parts.push("Http"); + if (fn.effects.clock) parts.push("Clock"); + if (fn.effects.random) parts.push("Random"); + if (fn.usesEnv) parts.push("env"); + return parts.length === 0 ? "Pure" : parts.join("+"); +} + +function dumpStmt(statement: CStmt, depth: number): string[] { + const pad = " ".repeat(depth); + switch (statement.kind) { + case "let": + return [`${pad}${statement.mutable ? "let" : "const"} ${statement.name}: ${typeToString(statement.type)} = ${dumpExpr(statement.init)}`]; + case "assign": + return [`${pad}${statement.name} = ${dumpExpr(statement.value)}`]; + case "setIndex": + return [`${pad}${statement.name}[${dumpExpr(statement.index)}] = ${dumpExpr(statement.value)}`]; + case "push": + return [`${pad}push ${statement.name} <- ${dumpExpr(statement.value)}`]; + case "if": + return [ + `${pad}if ${dumpExpr(statement.test)} {`, + ...statement.then.flatMap((item) => dumpStmt(item, depth + 1)), + ...(statement.otherwise.length === 0 + ? [] + : [`${pad}} else {`, ...statement.otherwise.flatMap((item) => dumpStmt(item, depth + 1))]), + `${pad}}`, + ]; + case "switch": + return [ + `${pad}switch ${dumpExpr(statement.subject)} {`, + ...statement.cases.flatMap((entry) => [ + `${pad} case ${entry.values.map((value) => JSON.stringify(value)).join(", ")}:`, + ...entry.body.flatMap((item) => dumpStmt(item, depth + 2)), + ]), + ...(statement.otherwise === undefined + ? [] + : [`${pad} default:`, ...statement.otherwise.flatMap((item) => dumpStmt(item, depth + 2))]), + `${pad}}`, + ]; + case "forRange": + return [ + `${pad}for ${statement.name}: ${typeToString(statement.type)} = ${dumpExpr(statement.from)} ${statement.step > 0n ? "to" : "downto"} ${dumpExpr(statement.to)} {`, + ...statement.body.flatMap((item) => dumpStmt(item, depth + 1)), + `${pad}}`, + ]; + case "forEach": + return [ + `${pad}forEach ${statement.name}: ${typeToString(statement.type)} in ${dumpExpr(statement.iterable)} {`, + ...statement.body.flatMap((item) => dumpStmt(item, depth + 1)), + `${pad}}`, + ]; + case "return": + return [`${pad}return${statement.value === undefined ? "" : ` ${dumpExpr(statement.value)}`}`]; + case "fail": + return [`${pad}fail ${statement.errorClass}(${statement.args.map(dumpExpr).join(", ")})`]; + case "break": + return [`${pad}break`]; + case "continue": + return [`${pad}continue`]; + case "expr": + return [`${pad}${dumpExpr(statement.expr)}`]; + default: { + const exhaustive: never = statement; + return exhaustive; + } + } +} + +export function dumpExpr(expr: CExpr): string { + switch (expr.kind) { + case "lit": + return `${literal(expr.value)}: ${typeToString(expr.type)}`; + case "local": + return `${expr.name}: ${typeToString(expr.type)}`; + case "none": + return "none"; + case "some": + return `some(${dumpExpr(expr.inner)})`; + case "record": + return `${expr.typeName} { ${expr.fields.map((field) => `${field.name}: ${dumpExpr(field.value)}`).join(", ")} }`; + case "field": + return `${dumpExpr(expr.target)}.${expr.name}`; + case "list": + return `[${expr.items.map(dumpExpr).join(", ")}]`; + case "call": + return `${expr.fn}(${expr.args.map(dumpExpr).join(", ")})`; + case "op": + return `${expr.op}${expr.regex === undefined ? "" : `/${expr.regex.source}/`}(${expr.args.map(dumpExpr).join(", ")}): ${typeToString(expr.type)}`; + case "lambda": + return `(${expr.params.map((param) => `${param.name}: ${typeToString(param.type)}`).join(", ")}) => …`; + case "cond": + return `(${dumpExpr(expr.test)} ? ${dumpExpr(expr.then)} : ${dumpExpr(expr.otherwise)})`; + case "and": + return `(${dumpExpr(expr.left)} && ${dumpExpr(expr.right)})`; + case "or": + return `(${dumpExpr(expr.left)} || ${dumpExpr(expr.right)})`; + case "not": + return `!${dumpExpr(expr.operand)}`; + default: { + const exhaustive: never = expr; + return exhaustive; + } + } +} + +function literal(value: Value): string { + if (typeof value === "bigint") return value.toString(); + if (typeof value === "string") return JSON.stringify(value); + if (Array.isArray(value)) return `[${value.map(literal).join(", ")}]`; + return String(value); +} diff --git a/engine/src/diagnostics.ts b/engine/src/diagnostics.ts new file mode 100644 index 000000000..80a7beb8a --- /dev/null +++ b/engine/src/diagnostics.ts @@ -0,0 +1,85 @@ +/** + * Span-aware diagnostics shared by every compiler stage. + * + * A diagnostic always names a stable code (`E_...`), so tests assert on the code rather than on + * the message text, and carries a span so the CLI can print the offending source line. + */ + +export type Span = { + readonly file: string; + readonly start: number; + readonly end: number; +}; + +export const NO_SPAN: Span = { file: "", start: 0, end: 0 }; + +export type Severity = "error" | "warning"; + +export type Diagnostic = { + readonly severity: Severity; + readonly code: string; + readonly message: string; + readonly span: Span; + /** What the author should write instead, when the compiler can tell. */ + readonly suggestion?: string; +}; + +/** A compilation failure carrying every diagnostic collected before the stage gave up. */ +export class CompileError extends Error { + readonly diagnostics: readonly Diagnostic[]; + + constructor(diagnostics: readonly Diagnostic[]) { + super(diagnostics.map((diagnostic) => `${diagnostic.code}: ${diagnostic.message}`).join("\n")); + this.name = "CompileError"; + this.diagnostics = diagnostics; + } +} + +export function error( + code: string, + message: string, + span: Span, + suggestion?: string, +): Diagnostic { + return { severity: "error", code, message, span, suggestion }; +} + +/** Collects diagnostics for one compilation and fails the stage on demand. */ +export class Diagnostics { + private readonly items: Diagnostic[] = []; + + add(diagnostic: Diagnostic): void { + this.items.push(diagnostic); + } + + error(code: string, message: string, span: Span, suggestion?: string): void { + this.add(error(code, message, span, suggestion)); + } + + get all(): readonly Diagnostic[] { + return this.items; + } + + get hasErrors(): boolean { + return this.items.some((item) => item.severity === "error"); + } + + throwIfErrors(): void { + if (this.hasErrors) throw new CompileError([...this.items]); + } +} + +/** Renders a diagnostic with its source line and a caret, for the CLI. */ +export function renderDiagnostic(diagnostic: Diagnostic, source?: string): string { + const head = `${diagnostic.severity} ${diagnostic.code}: ${diagnostic.message}`; + if (source === undefined) return `${head}\n at ${diagnostic.span.file}`; + + const before = source.slice(0, diagnostic.span.start); + const line = before.split("\n").length; + const column = diagnostic.span.start - (before.lastIndexOf("\n") + 1); + const text = source.split("\n")[line - 1] ?? ""; + const caret = `${" ".repeat(column)}${"^".repeat(Math.max(1, Math.min(diagnostic.span.end - diagnostic.span.start, text.length - column)))}`; + const suggestion = diagnostic.suggestion === undefined ? "" : `\n help: ${diagnostic.suggestion}`; + + return `${head}\n at ${diagnostic.span.file}:${line}:${column + 1}\n ${text}\n ${caret}${suggestion}`; +} diff --git a/engine/src/effects.ts b/engine/src/effects.ts new file mode 100644 index 000000000..aaa5a1ee4 --- /dev/null +++ b/engine/src/effects.ts @@ -0,0 +1,67 @@ +/** + * Effects and capabilities. + * + * `Fail` records the domain errors a function may raise; `Http`, `Clock` and `Random` are + * capabilities. The compiler infers all four over the call graph, and threads a capability record + * into exactly the functions that transitively need one. + */ + +export type EffectSet = { + /** Names of the domain error types the operation may raise. */ + readonly fail: readonly string[]; + readonly http: boolean; + readonly clock: boolean; + readonly random: boolean; +}; + +export const PURE: EffectSet = { fail: [], http: false, clock: false, random: false }; + +export function effects(partial: Partial): EffectSet { + return { ...PURE, ...partial }; +} + +export function unionEffects(...sets: readonly EffectSet[]): EffectSet { + const fail = new Set(); + let http = false; + let clock = false; + let random = false; + for (const set of sets) { + for (const name of set.fail) fail.add(name); + http ||= set.http; + clock ||= set.clock; + random ||= set.random; + } + return { fail: [...fail].sort(), http, clock, random }; +} + +export function withFail(set: EffectSet, errorType: string): EffectSet { + return unionEffects(set, { fail: [errorType], http: false, clock: false, random: false }); +} + +/** True when the operation needs the capability record threaded into it. */ +export function needsEnv(set: EffectSet): boolean { + return set.http || set.clock || set.random; +} + +export function isPure(set: EffectSet): boolean { + return set.fail.length === 0 && !needsEnv(set); +} + +export function effectsToString(set: EffectSet): string { + const parts: string[] = []; + if (set.fail.length > 0) parts.push(...set.fail.map((name) => `Fail<${name}>`)); + if (set.http) parts.push("Http"); + if (set.clock) parts.push("Clock"); + if (set.random) parts.push("Random"); + return parts.length === 0 ? "Pure" : parts.join(" + "); +} + +/** True when `a` is allowed wherever `b` is expected. */ +export function effectsSubsumed(a: EffectSet, b: EffectSet): boolean { + return ( + a.fail.every((name) => b.fail.includes(name)) && + (!a.http || b.http) && + (!a.clock || b.clock) && + (!a.random || b.random) + ); +} diff --git a/engine/src/frontend/lower.ts b/engine/src/frontend/lower.ts new file mode 100644 index 000000000..d263341ae --- /dev/null +++ b/engine/src/frontend/lower.ts @@ -0,0 +1,995 @@ +/** + * The TypeScript frontend: `oxc-parser` output in, Semantic HIR out. + * + * Everything the subset rejects (docs/semantics.md, "Subset") is rejected here, with a span and, + * where the compiler can tell, the construct to write instead. Nothing downstream ever sees a + * TypeScript node: this module is the only one in the engine that imports the parser. + */ + +import { parseSync } from "oxc-parser"; +import type { Diagnostics, Span } from "../diagnostics.ts"; +import type { + HConst, + HErrorDecl, + HExpr, + HFunc, + HImport, + HModule, + HParam, + HStmt, + HTypeDecl, + HTypeExpr, +} from "../hir/ast.ts"; + +/** The untyped ESTree-shaped JSON the parser hands back. */ +type Node = any; + +const HOST_GLOBALS = new Set([ + "Date", + "Math", + "JSON", + "Intl", + "fetch", + "console", + "setTimeout", + "setInterval", + "Promise", + "RegExp", + "Number", + "String", + "Boolean", + "Object", + "Array", + "Map", + "Set", + "BigInt", + "globalThis", + "process", + "structuredClone", +]); + +const HOST_GLOBAL_HELP: Record = { + Date: "civil dates come from the `date` intrinsics; a host date is the DX's job", + Math: "write the arithmetic in source, or use an admitted intrinsic", + JSON: "parse in source; there is no host JSON in the core", + Intl: "formatting is source library code, so every target formats identically", + fetch: "call `http.request`; the capability is threaded for you", + console: "the core has no output side", + Number: "use `str.parseInt` or `float.fromInt`", + RegExp: "write a regex literal; patterns are compile-time only", +}; + +/** + * `Math` methods with a sound, always-Int lowering: every call site the source needs today is + * integer arithmetic, and there is no `float.*` counterpart yet (admitting one needs a second + * caller, docs/semantics.md "Admission rule for intrinsics"), so the checker rejects a Float + * operand instead of silently truncating it. + */ +const MATH_METHODS = new Set(["min", "max", "abs", "trunc", "floor"]); + +export function parseModule( + file: string, + path: string, + source: string, + diagnostics: Diagnostics, +): HModule { + const parsed = parseSync(file, source, { lang: "ts" }); + for (const parseError of parsed.errors) { + diagnostics.error("E_PARSE", parseError.message, { + file, + start: parseError.labels?.[0]?.start ?? 0, + end: parseError.labels?.[0]?.end ?? 0, + }); + } + diagnostics.throwIfErrors(); + return new Lowering(file, path, source, diagnostics).module(parsed.program); +} + +class Lowering { + private readonly file: string; + private readonly path: string; + private readonly source: string; + private readonly diagnostics: Diagnostics; + + constructor(file: string, path: string, source: string, diagnostics: Diagnostics) { + this.file = file; + this.path = path; + this.source = source; + this.diagnostics = diagnostics; + } + + private span(node: Node): Span { + return { file: this.file, start: node?.start ?? 0, end: node?.end ?? 0 }; + } + + private reject(code: string, message: string, node: Node, suggestion?: string): void { + this.diagnostics.error(code, message, this.span(node), suggestion); + } + + private doc(node: Node): string | undefined { + // The JSDoc block immediately above the declaration, kept for the generated code's header. + const before = this.source.slice(0, node.start); + const match = /\/\*\*((?:[^*]|\*(?!\/))*)\*\/\s*(?:export\s+)?$/.exec(before); + if (match === null) return undefined; + return match[1]! + .split("\n") + .map((line) => line.replace(/^\s*\*ic?/, "").replace(/^\s*\*\s?/, "").trimEnd()) + .join("\n") + .trim(); + } + + module(program: Node): HModule { + const imports: HImport[] = []; + const types: HTypeDecl[] = []; + const errors: HErrorDecl[] = []; + const consts: HConst[] = []; + const functions: HFunc[] = []; + + for (const statement of program.body) { + this.topLevel(statement, false, { imports, types, errors, consts, functions }); + } + + return { + path: this.path, + file: this.file, + source: this.source, + imports, + types, + errors, + consts, + functions, + }; + } + + private topLevel( + node: Node, + exported: boolean, + out: { + imports: HImport[]; + types: HTypeDecl[]; + errors: HErrorDecl[]; + consts: HConst[]; + functions: HFunc[]; + }, + ): void { + switch (node.type) { + case "ImportDeclaration": { + if (node.importKind === "type") return; + out.imports.push({ + from: node.source.value, + names: (node.specifiers ?? []) + .filter((specifier: Node) => specifier.type === "ImportSpecifier") + .map((specifier: Node) => ({ + imported: specifier.imported.name, + local: specifier.local.name, + })), + span: this.span(node), + }); + return; + } + case "ExportNamedDeclaration": { + if (node.declaration === null || node.declaration === undefined) return; + this.topLevel(node.declaration, true, out); + return; + } + case "ExportDefaultDeclaration": + this.reject("E_DEFAULT_EXPORT", "default exports are outside the subset", node, "use a named export"); + return; + case "TSTypeAliasDeclaration": { + out.types.push({ + name: node.id.name, + type: this.typeExpr(node.typeAnnotation), + exported, + doc: this.doc(node), + span: this.span(node), + }); + return; + } + case "TSInterfaceDeclaration": + this.reject("E_INTERFACE", "interfaces are outside the subset", node, "declare a `type` alias instead"); + return; + case "TSEnumDeclaration": + this.reject( + "E_TS_ENUM", + "TypeScript enums are outside the subset", + node, + 'use a string literal union, for example `type Version = "v1" | "v2"`', + ); + return; + case "ClassDeclaration": { + if (node.body.body.length > 0) { + this.reject( + "E_CLASS_BODY", + "only empty error classes are allowed", + node, + "move the behavior into a function", + ); + } + if (node.superClass === null || node.superClass === undefined) { + this.reject("E_CLASS_BASE", "a class must extend an error type", node); + return; + } + out.errors.push({ + name: node.id.name, + base: node.superClass.name, + exported, + doc: this.doc(node), + span: this.span(node), + }); + return; + } + case "VariableDeclaration": { + if (node.kind !== "const") { + this.reject("E_TOP_LEVEL_LET", "module level bindings must be `const`", node); + } + for (const declarator of node.declarations) { + if (declarator.id.type !== "Identifier") { + this.reject("E_DESTRUCTURING", "destructuring is outside the subset", declarator); + continue; + } + const init = declarator.init; + if (init === null || init === undefined) { + this.reject("E_UNINITIALIZED", "a constant must be initialized", declarator); + continue; + } + if (init.type === "ArrowFunctionExpression" || init.type === "FunctionExpression") { + out.functions.push(this.functionFromArrow(node, declarator, init, exported)); + continue; + } + out.consts.push({ + name: declarator.id.name, + declared: + declarator.id.typeAnnotation === null || declarator.id.typeAnnotation === undefined + ? undefined + : this.typeExpr(declarator.id.typeAnnotation.typeAnnotation), + value: this.expr(init), + exported, + span: this.span(declarator), + }); + } + return; + } + case "FunctionDeclaration": { + out.functions.push({ + name: node.id.name, + params: this.params(node.params), + ret: this.returnType(node), + body: this.block(node.body), + exported, + doc: this.doc(node), + span: this.span(node), + }); + return; + } + case "TSModuleDeclaration": + this.reject("E_NAMESPACE", "namespaces are outside the subset", node); + return; + default: + this.reject("E_TOP_LEVEL", `${node.type} is not allowed at module level`, node); + } + } + + private functionFromArrow(declaration: Node, declarator: Node, arrow: Node, exported: boolean): HFunc { + if (arrow.async === true) { + this.reject( + "E_ASYNC", + "`async` is computed by the compiler, not written", + arrow, + "call `http.request` directly; the TypeScript backend adds async where it is needed", + ); + } + return { + name: declarator.id.name, + params: this.params(arrow.params), + ret: this.returnType(arrow), + body: arrow.body.type === "BlockStatement" ? this.block(arrow.body) : [ + { kind: "return", value: this.expr(arrow.body), span: this.span(arrow.body) }, + ], + exported, + doc: this.doc(declaration), + span: this.span(declarator), + }; + } + + private returnType(node: Node): HTypeExpr { + if (node.returnType === null || node.returnType === undefined) { + this.reject( + "E_MISSING_RETURN_TYPE", + "every function needs an explicit return type", + node, + "annotate the return type; the core's signatures are a published contract", + ); + return { kind: "ref", name: "Void", args: [], span: this.span(node) }; + } + return this.typeExpr(node.returnType.typeAnnotation); + } + + private params(params: readonly Node[]): HParam[] { + const result: HParam[] = []; + for (const param of params) { + if (param.type !== "Identifier") { + this.reject("E_PARAM_PATTERN", "parameter patterns are outside the subset", param); + continue; + } + if (param.typeAnnotation === null || param.typeAnnotation === undefined) { + this.reject("E_MISSING_PARAM_TYPE", `parameter \`${param.name}\` needs a type`, param); + continue; + } + result.push({ + name: param.name, + type: this.typeExpr(param.typeAnnotation.typeAnnotation), + span: this.span(param), + }); + } + return result; + } + + /* ---------------------------------------------------------------- * + * Types + * ---------------------------------------------------------------- */ + + private typeExpr(node: Node): HTypeExpr { + const span = this.span(node); + switch (node.type) { + case "TSNumberKeyword": + // Ordinary TypeScript, accepted as written: the checker infers what range this stands + // for, from a guard the body writes (an exported utility) or from each call site + // (library code, the same specialization `Int` already gets) — see check.ts, "Inferring + // a bare `number`". `Int`, `IntRange`, `Float` and `Decimal` are still + // there for an author who wants to state the contract explicitly, and `E_BARE_NUMBER` + // still fires, later, for a parameter no guard and no caller ever narrows. + return { kind: "ref", name: "number", args: [], span }; + case "TSAnyKeyword": + case "TSUnknownKeyword": + this.reject("E_ANY", "`any` and `unknown` are outside the subset", node); + return { kind: "ref", name: "Int", args: [], span }; + case "TSNullKeyword": + this.reject("E_NULL", "`null` is outside the subset", node, "use `T | undefined`"); + return { kind: "undefined", span }; + case "TSStringKeyword": + return { kind: "ref", name: "String", args: [], span }; + case "TSBooleanKeyword": + return { kind: "ref", name: "Bool", args: [], span }; + case "TSVoidKeyword": + return { kind: "ref", name: "Void", args: [], span }; + case "TSUndefinedKeyword": + return { kind: "undefined", span }; + case "TSTypeReference": { + const name = node.typeName.type === "Identifier" ? node.typeName.name : ""; + return { + kind: "ref", + name, + args: (node.typeArguments?.params ?? []).map((arg: Node) => this.typeExpr(arg)), + span, + }; + } + case "TSArrayType": + return { kind: "array", elem: this.typeExpr(node.elementType), span }; + case "TSTypeOperator": + // `readonly T[]`: every value is immutable in the semantics, so the operator is noise. + return this.typeExpr(node.typeAnnotation); + case "TSUnionType": + return { kind: "union", options: node.types.map((item: Node) => this.typeExpr(item)), span }; + case "TSLiteralType": { + const literal = node.literal; + if (literal.type === "Literal" && typeof literal.value === "string") { + return { kind: "literal", value: literal.value, span }; + } + if (literal.type === "Literal" && typeof literal.value === "number") { + return { kind: "literal", value: String(literal.value), span }; + } + if ( + literal.type === "UnaryExpression" && + literal.operator === "-" && + literal.argument.type === "Literal" + ) { + return { kind: "literal", value: `-${literal.argument.value}`, span }; + } + this.reject("E_LITERAL_TYPE", "only string and integer literal types are supported", node); + return { kind: "literal", value: "", span }; + } + case "TSTypeLiteral": + return { + kind: "object", + fields: node.members.map((member: Node) => { + if (member.type !== "TSPropertySignature") { + this.reject("E_TYPE_MEMBER", "only properties are allowed in a record type", member); + } + return { + name: member.key.name ?? member.key.value, + type: + member.typeAnnotation === null || member.typeAnnotation === undefined + ? { kind: "ref" as const, name: "Int", args: [], span: this.span(member) } + : this.typeExpr(member.typeAnnotation.typeAnnotation), + optional: member.optional === true, + doc: this.doc(member), + span: this.span(member), + }; + }), + span, + }; + case "TSFunctionType": + return { + kind: "func", + params: this.params(node.params).map((param) => param.type), + ret: this.typeExpr(node.returnType.typeAnnotation), + span, + }; + case "TSParenthesizedType": + return this.typeExpr(node.typeAnnotation); + default: + this.reject("E_TYPE", `${node.type} is not a supported type`, node); + return { kind: "ref", name: "Int", args: [], span }; + } + } + + /* ---------------------------------------------------------------- * + * Statements + * ---------------------------------------------------------------- */ + + private block(node: Node): HStmt[] { + return (node.body ?? []).flatMap((statement: Node) => this.stmt(statement)); + } + + private stmt(node: Node): HStmt[] { + const span = this.span(node); + switch (node.type) { + case "VariableDeclaration": { + if (node.kind === "var") { + this.reject("E_VAR", "`var` is outside the subset", node, "use `const` or `let`"); + } + const result: HStmt[] = []; + for (const declarator of node.declarations) { + if (declarator.id.type !== "Identifier") { + this.reject("E_DESTRUCTURING", "destructuring is outside the subset", declarator); + continue; + } + if (declarator.init === null || declarator.init === undefined) { + this.reject("E_UNINITIALIZED", "a binding must be initialized", declarator); + continue; + } + result.push({ + kind: "let", + name: declarator.id.name, + mutable: node.kind === "let", + declared: + declarator.id.typeAnnotation === null || declarator.id.typeAnnotation === undefined + ? undefined + : this.typeExpr(declarator.id.typeAnnotation.typeAnnotation), + init: this.expr(declarator.init), + span: this.span(declarator), + }); + } + return result; + } + case "ExpressionStatement": { + const expression = node.expression; + if (expression.type === "AssignmentExpression") { + const operator = expression.operator; + if (!["=", "+=", "-=", "*="].includes(operator)) { + this.reject("E_ASSIGN_OP", `\`${operator}\` is outside the subset`, expression); + return []; + } + return [ + { + kind: "assign", + target: this.expr(expression.left), + op: operator as "=" | "+=" | "-=" | "*=", + value: this.expr(expression.right), + span, + }, + ]; + } + if (expression.type === "UpdateExpression") { + return [ + { + kind: "assign", + target: this.expr(expression.argument), + op: expression.operator === "++" ? "+=" : "-=", + value: { kind: "int", value: 1n, span }, + span, + }, + ]; + } + return [{ kind: "expr", expr: this.expr(expression), span }]; + } + case "IfStatement": + return [ + { + kind: "if", + test: this.expr(node.test), + then: this.bodyOf(node.consequent), + otherwise: + node.alternate === null || node.alternate === undefined + ? undefined + : this.bodyOf(node.alternate), + span, + }, + ]; + case "SwitchStatement": + return [ + { + kind: "switch", + subject: this.expr(node.discriminant), + cases: node.cases.map((caseNode: Node) => ({ + test: + caseNode.test === null || caseNode.test === undefined + ? undefined + : this.expr(caseNode.test), + body: caseNode.consequent.flatMap((item: Node) => this.stmt(item)), + span: this.span(caseNode), + })), + span, + }, + ]; + case "ForStatement": + return this.countedFor(node); + case "ForOfStatement": { + if (node.left.type !== "VariableDeclaration" || node.left.declarations.length !== 1) { + this.reject("E_FOR_OF", "`for…of` binds exactly one name", node); + return []; + } + return [ + { + kind: "forOf", + name: node.left.declarations[0].id.name, + iterable: this.expr(node.right), + body: this.bodyOf(node.body), + span, + }, + ]; + } + case "ForInStatement": + this.reject("E_FOR_IN", "`for…in` is outside the subset", node, "iterate a list with `for…of`"); + return []; + case "WhileStatement": + case "DoWhileStatement": + this.reject( + "E_WHILE", + "unbounded loops are outside the subset", + node, + "use a counted `for` or `for…of`, so every loop has a proven trip count", + ); + return []; + case "ReturnStatement": + return [ + { + kind: "return", + value: + node.argument === null || node.argument === undefined + ? undefined + : this.expr(node.argument), + span, + }, + ]; + case "ThrowStatement": { + if (node.argument.type !== "NewExpression" || node.argument.callee.type !== "Identifier") { + this.reject( + "E_THROW", + "only `throw new SomeError(...)` is allowed", + node, + "declare the error class in source and throw it directly", + ); + return []; + } + return [ + { + kind: "throw", + errorClass: node.argument.callee.name, + args: node.argument.arguments.map((arg: Node) => this.expr(arg)), + span, + }, + ]; + } + case "TryStatement": + this.reject( + "E_TRY", + "`try`/`catch` is outside the subset", + node, + "return an Option or let the domain error propagate; bugs are proven impossible instead of caught", + ); + return []; + case "BreakStatement": + return [{ kind: "break", span }]; + case "ContinueStatement": + return [{ kind: "continue", span }]; + case "BlockStatement": + return [{ kind: "block", body: this.block(node), span }]; + case "EmptyStatement": + return []; + default: + this.reject("E_STATEMENT", `${node.type} is outside the subset`, node); + return []; + } + } + + private bodyOf(node: Node): HStmt[] { + return node.type === "BlockStatement" ? this.block(node) : this.stmt(node); + } + + /** Accepts only the recognizable counted shape `for (let i = a; i < b; i++)`. */ + private countedFor(node: Node): HStmt[] { + const span = this.span(node); + const init = node.init; + const test = node.test; + const update = node.update; + const shape = + init?.type === "VariableDeclaration" && + init.declarations.length === 1 && + init.declarations[0].id.type === "Identifier" && + test?.type === "BinaryExpression" && + ["<", "<=", ">", ">="].includes(test.operator) && + test.left.type === "Identifier" && + test.left.name === init.declarations[0].id.name && + update?.type === "UpdateExpression" && + update.argument.type === "Identifier" && + update.argument.name === init.declarations[0].id.name; + + if (!shape) { + this.reject( + "E_FOR_SHAPE", + "only counted `for (let i = a; i < b; i++)` loops are allowed", + node, + "rewrite as a counted loop or as `for…of`, so the trip count is proven", + ); + return []; + } + + const descending = test.operator === ">" || test.operator === ">="; + if (descending !== (update.operator === "--")) { + this.reject("E_FOR_DIRECTION", "the loop counter moves away from its bound", node); + } + + return [ + { + kind: "forCounted", + name: init.declarations[0].id.name, + from: this.expr(init.declarations[0].init), + to: this.expr(test.right), + inclusive: test.operator === "<=" || test.operator === ">=", + step: update.operator === "++" ? 1n : -1n, + body: this.bodyOf(node.body), + span, + }, + ]; + } + + /* ---------------------------------------------------------------- * + * Expressions + * ---------------------------------------------------------------- */ + + /** + * Recognizes the handful of ordinary-JavaScript call shapes that have no vocabulary of their + * own in this subset — `Math.min`/`max`/`abs`/`trunc`/`floor`, `Math.random`, and `String(n)` + * — before the generic `Identifier` lowering below would reject their host global outright, + * with a less specific message than each of these deserves. Returns `undefined` for any other + * call, which falls through to that generic lowering unchanged. + */ + private specialCall(node: Node, span: Span): HExpr | undefined { + const callee = node.callee; + if ( + callee.type === "MemberExpression" && + callee.computed === false && + callee.object.type === "Identifier" && + callee.object.name === "Math" + ) { + const method = callee.property.name as string; + if (method === "random") { + this.reject( + "E_MATH_RANDOM", + "`Math.random()` is a float in [0, 1); the Random capability offers only " + + "`random.nextU32()`, an unbiased 32-bit draw", + node, + "call `random.nextU32()` and derive what you need from it in source, the way " + + "`lib/random.ts` does — a scaled float would have to round identically in every " + + "target to stay unbiased, so there is no built-in shortcut", + ); + return { kind: "float", value: 0, span }; + } + if (MATH_METHODS.has(method)) { + return { + kind: "call", + callee: { + kind: "member", + target: { kind: "name", name: "Math", span: this.span(callee.object) }, + name: method, + span: this.span(callee), + }, + args: node.arguments.map((arg: Node) => this.expr(arg)), + span, + }; + } + return undefined; + } + if (callee.type === "Identifier" && callee.name === "String" && node.arguments.length === 1) { + return { + kind: "call", + callee: { kind: "name", name: "String", span: this.span(callee) }, + args: [this.expr(node.arguments[0])], + span, + }; + } + // `value[index]?.charCodeAt(0)` is `str.codeAtOpt(value, index)`: the one ordinary spelling + // of the checked *numeric* accessor. `value.charCodeAt(i)` alone always answers `NaN` past + // the end, not `undefined`, so it has no `??` form (see the `logical` handling of `xs[i] ?? + // fallback`); but `value[index]` alone already answers `undefined` there, and chaining + // `?.charCodeAt(0)` onto it reads the one scalar's code point only when it is present — the + // same case split `str.codeAtOpt` makes, spelled in ordinary TypeScript instead of assumed. + // The literal `0` and the plain (non-optional) bracket index are both required: anything + // else is not this idiom and falls through to the generic `?.` rejection below, whose + // message points back here. + if ( + callee.type === "MemberExpression" && + callee.optional === true && + callee.computed === false && + callee.property.name === "charCodeAt" && + callee.object.type === "MemberExpression" && + callee.object.computed === true && + callee.object.optional !== true && + node.arguments.length === 1 && + node.arguments[0].type === "Literal" && + node.arguments[0].value === 0 + ) { + return { + kind: "call", + callee: { + kind: "member", + target: { kind: "name", name: "str", span: this.span(callee.object) }, + name: "codeAtOpt", + span: this.span(callee), + }, + args: [this.expr(callee.object.object), this.expr(callee.object.property)], + span, + }; + } + return undefined; + } + + private expr(node: Node): HExpr { + const span = this.span(node); + switch (node.type) { + case "Literal": { + if (node.regex !== null && node.regex !== undefined) { + return { kind: "regex", source: node.regex.pattern, flags: node.regex.flags, span }; + } + if (node.value === null) { + this.reject("E_NULL", "`null` is outside the subset", node, "use `undefined`"); + return { kind: "undefined", span }; + } + if (typeof node.value === "boolean") return { kind: "bool", value: node.value, span }; + if (typeof node.value === "string") return { kind: "string", value: node.value, span }; + if (typeof node.value === "number") { + return Number.isInteger(node.value) && !node.raw.includes(".") && !node.raw.includes("e") + ? { kind: "int", value: BigInt(node.raw.replaceAll("_", "")), span } + : { kind: "float", value: node.value, span }; + } + if (typeof node.value === "bigint") return { kind: "int", value: node.value, span }; + this.reject("E_LITERAL", "unsupported literal", node); + return { kind: "undefined", span }; + } + case "Identifier": { + if (node.name === "undefined") return { kind: "undefined", span }; + if (HOST_GLOBALS.has(node.name)) { + this.reject( + "E_HOST_GLOBAL", + `the host global \`${node.name}\` is outside the subset`, + node, + HOST_GLOBAL_HELP[node.name], + ); + } + return { kind: "name", name: node.name, span }; + } + case "ThisExpression": + this.reject("E_THIS", "`this` is outside the subset", node); + return { kind: "undefined", span }; + case "MemberExpression": { + if (node.optional === true) { + this.reject( + "E_OPTIONAL_CHAIN", + "`?.` is allowed only on an Option, which the checker narrows explicitly", + node, + node.property?.name === "charCodeAt" + ? "the checked numeric accessor has exactly one ordinary spelling: " + + "`value[index]?.charCodeAt(0)`, a plain (non-optional) bracket index and a " + + "literal `0` — anything else, including a variable in place of the `0`, is not " + + "this idiom and has no honest translation, since `value.charCodeAt(i)` alone " + + "answers `NaN` past the end, not `undefined`" + : "check for `undefined` first", + ); + } + if (node.computed === true) { + return { kind: "index", target: this.expr(node.object), index: this.expr(node.property), span }; + } + return { kind: "member", target: this.expr(node.object), name: node.property.name, span }; + } + case "CallExpression": { + if (node.optional === true) { + this.reject("E_OPTIONAL_CALL", "`?.()` is outside the subset", node); + } + const special = this.specialCall(node, span); + if (special !== undefined) return special; + return { + kind: "call", + callee: this.expr(node.callee), + args: node.arguments.map((arg: Node) => { + if (arg.type === "SpreadElement") { + this.reject("E_SPREAD", "spread arguments are outside the subset", arg); + return this.expr(arg.argument); + } + return this.expr(arg); + }), + span, + }; + } + case "NewExpression": { + if (node.callee.name === "Date") { + this.reject( + "E_HOST_DATE", + "`new Date(...)` is JavaScript's own host object: months are zero-indexed, an " + + "out-of-range component silently rolls over into the next one, and the value is " + + "bound to a timezone, none of which a civil date does", + node, + "call `date.fromYmd(year, month, day)` (month is 1-12, and it answers `undefined` " + + "instead of rolling over) and handle the `undefined` case explicitly", + ); + return { kind: "undefined", span }; + } + return { + kind: "new", + className: node.callee.name, + args: node.arguments.map((arg: Node) => this.expr(arg)), + span, + }; + } + case "BinaryExpression": { + const operator = node.operator; + if (operator === "==" || operator === "!=") { + this.reject( + "E_LOOSE_EQUALITY", + "`==` is outside the subset", + node, + "use `===`, which the Core lowers to structural equality", + ); + return { kind: "bool", value: false, span }; + } + if (!["+", "-", "*", "/", "%", "<", "<=", ">", ">=", "===", "!=="].includes(operator)) { + this.reject("E_OPERATOR", `\`${operator}\` is outside the subset`, node); + return { kind: "bool", value: false, span }; + } + return { + kind: "binary", + op: operator, + left: this.expr(node.left), + right: this.expr(node.right), + span, + }; + } + case "LogicalExpression": + return { + kind: "logical", + op: node.operator, + left: this.expr(node.left), + right: this.expr(node.right), + span, + }; + case "UnaryExpression": { + if (node.operator !== "!" && node.operator !== "-") { + this.reject("E_UNARY", `unary \`${node.operator}\` is outside the subset`, node); + return { kind: "bool", value: false, span }; + } + return { kind: "unary", op: node.operator, operand: this.expr(node.argument), span }; + } + case "ConditionalExpression": + return { + kind: "ternary", + test: this.expr(node.test), + then: this.expr(node.consequent), + otherwise: this.expr(node.alternate), + span, + }; + case "TemplateLiteral": { + const parts: HExpr[] = node.expressions.map((item: Node) => this.expr(item)); + const merged: ({ kind: "text"; value: string } | { kind: "expr"; expr: HExpr })[] = []; + node.quasis.forEach((quasi: Node, index: number) => { + if (quasi.value.cooked !== "") merged.push({ kind: "text", value: quasi.value.cooked }); + const expression = parts[index]; + if (expression !== undefined) merged.push({ kind: "expr", expr: expression }); + }); + return { kind: "template", parts: merged, span }; + } + case "ObjectExpression": + return { + kind: "object", + fields: node.properties.map((property: Node) => { + if (property.type !== "Property" || property.computed === true) { + this.reject("E_OBJECT_PROPERTY", "only plain properties are allowed", property); + return { name: "", value: { kind: "undefined" as const, span }, span }; + } + return { + name: property.key.name ?? property.key.value, + value: this.expr(property.value), + span: this.span(property), + }; + }), + span, + }; + case "ArrayExpression": { + const elements: Node[] = node.elements; + if (elements.length === 1 && elements[0]?.type === "SpreadElement") { + // `[...s]`: JavaScript's array-spread of a string decomposes it into its scalars, + // which is exactly `str.codePoints`. + return { + kind: "call", + callee: { kind: "member", target: { kind: "name", name: "str", span }, name: "codePoints", span }, + args: [this.expr(elements[0].argument)], + span, + }; + } + if (elements.some((item: Node) => item?.type === "SpreadElement")) { + this.reject( + "E_ARRAY_SPREAD", + "array spread is outside the subset except for `[...s]` on a single string", + node, + "combine lists with `seq.concat`, and write anything else as an explicit loop or a combinator", + ); + } + return { + kind: "array", + items: elements.map((item: Node) => + item?.type === "SpreadElement" ? this.expr(item.argument) : this.expr(item), + ), + span, + }; + } + case "ArrowFunctionExpression": { + if (node.async === true) this.reject("E_ASYNC", "`async` is computed, not written", node); + return { + kind: "lambda", + params: node.params.map((param: Node) => ({ + name: param.name, + type: + param.typeAnnotation === null || param.typeAnnotation === undefined + ? undefined + : this.typeExpr(param.typeAnnotation.typeAnnotation), + span: this.span(param), + })), + body: + node.body.type === "BlockStatement" + ? this.block(node.body) + : [{ kind: "return", value: this.expr(node.body), span: this.span(node.body) }], + span, + }; + } + case "TSAsExpression": + case "TSSatisfiesExpression": + // `as const` on data literals is the only cast the subset keeps, and it is a no-op here. + return this.expr(node.expression); + case "TSNonNullExpression": + this.reject( + "E_NON_NULL", + "`!` does not prove anything to the checker", + node, + "narrow the Option with an explicit `=== undefined` check", + ); + return this.expr(node.expression); + case "ParenthesizedExpression": + return this.expr(node.expression); + case "ChainExpression": + // The parser wraps any expression containing a `?.` in this node, however deep — + // `value[index]?.charCodeAt(0)` arrives as `ChainExpression(CallExpression(…))`, not + // as the `CallExpression` directly. Unwrapping it here is what lets the `?.` shape + // recognized in `specialCall` (and the generic `E_OPTIONAL_CHAIN` rejection for + // every other one) ever see the node they match against. + return this.expr(node.expression); + case "AwaitExpression": + this.reject("E_AWAIT", "`await` is computed by the compiler, not written", node); + return this.expr(node.argument); + case "FunctionExpression": + this.reject("E_FUNCTION_EXPRESSION", "use an arrow function", node); + return { kind: "undefined", span }; + default: + this.reject("E_EXPRESSION", `${node.type} is outside the subset`, node); + return { kind: "undefined", span }; + } + } +} diff --git a/engine/src/fuzz/ast.ts b/engine/src/fuzz/ast.ts new file mode 100644 index 000000000..acb99d298 --- /dev/null +++ b/engine/src/fuzz/ast.ts @@ -0,0 +1,140 @@ +/** + * The generator's own tiny AST for a fuzzed function body, and its printer to source text. + * + * This is deliberately not the engine's HIR: the generator has to build and mutate (for + * shrinking) well-typed programs *before* they exist as text, and doing that against a + * purpose-built, minimal tree is far simpler than driving `oxc-parser`'s TypeScript grammar in + * reverse. The printer's only job is to emit source the engine's own frontend parses back into + * the same HIR shapes documented in `docs/semantics.md` section 7. + */ + +export type BinOp = "+" | "-" | "*" | "/" | "%" | "<" | "<=" | ">" | ">=" | "===" | "!=="; + +export type Expr = + | { readonly k: "int"; value: bigint } + | { readonly k: "bool"; value: boolean } + | { readonly k: "var"; name: string } + | { readonly k: "bin"; op: BinOp; l: Expr; r: Expr } + | { readonly k: "logical"; op: "&&" | "||"; l: Expr; r: Expr } + | { readonly k: "not"; e: Expr } + | { readonly k: "ternary"; test: Expr; then: Expr; else_: Expr } + /** `target[index]`, which the frontend lowers to `seq.get`/`str.charAt`. */ + | { readonly k: "index"; target: Expr; index: Expr } + /** `target.length`, the only member access the generator needs. */ + | { readonly k: "length"; target: Expr } + /** An intrinsic module call, `fn(...args)` (`str.codePoints`, `seq.at`, `str.fromCodePoints`, …). */ + | { readonly k: "call"; fn: string; args: readonly Expr[] } + /** An array literal; the generator only ever needs the empty one, `[]`. */ + | { readonly k: "emptyArray" }; + +export type Stmt = + | { readonly k: "let"; name: string; typeAnn: string; init: Expr; mutable: boolean } + | { readonly k: "assign"; name: string; op: "=" | "+=" | "-=" | "*="; value: Expr } + | { readonly k: "push"; name: string; value: Expr } + | { readonly k: "if"; test: Expr; then: readonly Stmt[]; else_?: readonly Stmt[] } + | { readonly k: "forCounted"; name: string; fromN: bigint; to: Expr; body: readonly Stmt[] } + | { readonly k: "forOf"; name: string; iterable: Expr; body: readonly Stmt[] } + | { + readonly k: "switch"; + subject: Expr; + cases: readonly { readonly test?: string; readonly body: readonly Stmt[] }[]; + } + | { readonly k: "return"; value?: Expr } + | { readonly k: "break" } + | { readonly k: "continue" }; + +export type Param = { readonly name: string; readonly typeAnn: string }; + +export type FuzzFunc = { + readonly enumDecl?: { readonly name: string; readonly members: readonly string[] }; + readonly params: readonly Param[]; + readonly retTypeAnn: string; + readonly body: readonly Stmt[]; + readonly fnName: string; +}; + +function printExpr(expr: Expr): string { + switch (expr.k) { + case "int": + return expr.value < 0n ? `(${expr.value})` : `${expr.value}`; + case "bool": + return `${expr.value}`; + case "var": + return expr.name; + case "bin": + return `(${printExpr(expr.l)} ${expr.op} ${printExpr(expr.r)})`; + case "logical": + return `(${printExpr(expr.l)} ${expr.op} ${printExpr(expr.r)})`; + case "not": + return `!(${printExpr(expr.e)})`; + case "ternary": + return `(${printExpr(expr.test)} ? ${printExpr(expr.then)} : ${printExpr(expr.else_)})`; + case "index": + return `${printExpr(expr.target)}[${printExpr(expr.index)}]`; + case "length": + return `${printExpr(expr.target)}.length`; + case "call": + return `${expr.fn}(${expr.args.map(printExpr).join(", ")})`; + case "emptyArray": + return "[]"; + } +} + +function printBlock(body: readonly Stmt[], indent: string): string { + if (body.length === 0) return `${indent}\n`; + return body.map((stmt) => printStmt(stmt, indent)).join(""); +} + +function printStmt(stmt: Stmt, indent: string): string { + switch (stmt.k) { + case "let": + return `${indent}${stmt.mutable ? "let" : "const"} ${stmt.name}: ${stmt.typeAnn} = ${printExpr(stmt.init)};\n`; + case "assign": + return `${indent}${stmt.name} ${stmt.op} ${printExpr(stmt.value)};\n`; + case "push": + return `${indent}${stmt.name}.push(${printExpr(stmt.value)});\n`; + case "if": { + const then = `${indent}if (${printExpr(stmt.test)}) {\n${printBlock(stmt.then, `${indent}\t`)}${indent}}`; + if (stmt.else_ === undefined) return `${then}\n`; + return `${then} else {\n${printBlock(stmt.else_, `${indent}\t`)}${indent}}\n`; + } + case "forCounted": + return ( + `${indent}for (let ${stmt.name} = ${stmt.fromN}; ${stmt.name} < ${printExpr(stmt.to)}; ${stmt.name}++) {\n` + + `${printBlock(stmt.body, `${indent}\t`)}${indent}}\n` + ); + case "forOf": + return ( + `${indent}for (const ${stmt.name} of ${printExpr(stmt.iterable)}) {\n` + + `${printBlock(stmt.body, `${indent}\t`)}${indent}}\n` + ); + case "switch": { + const cases = stmt.cases + .map((c) => { + const label = c.test === undefined ? `${indent}\tdefault:\n` : `${indent}\tcase ${c.test}:\n`; + return `${label}${printBlock(c.body, `${indent}\t\t`)}`; + }) + .join(""); + return `${indent}switch (${printExpr(stmt.subject)}) {\n${cases}${indent}}\n`; + } + case "return": + return `${indent}return${stmt.value === undefined ? "" : ` ${printExpr(stmt.value)}`};\n`; + case "break": + return `${indent}break;\n`; + case "continue": + return `${indent}continue;\n`; + } +} + +/** Renders a whole module: the optional enum, and the one exported entry point. */ +export function printProgram(fn: FuzzFunc): string { + const lines: string[] = []; + if (fn.enumDecl !== undefined) { + lines.push(`export type ${fn.enumDecl.name} = ${fn.enumDecl.members.map((m) => `"${m}"`).join(" | ")};\n\n`); + } + const params = fn.params.map((p) => `${p.name}: ${p.typeAnn}`).join(", "); + lines.push(`export function ${fn.fnName}(${params}): ${fn.retTypeAnn} {\n`); + lines.push(printBlock(fn.body, "\t")); + lines.push(`}\n`); + return lines.join(""); +} diff --git a/engine/src/fuzz/generate.ts b/engine/src/fuzz/generate.ts new file mode 100644 index 000000000..1816fc52d --- /dev/null +++ b/engine/src/fuzz/generate.ts @@ -0,0 +1,402 @@ +/** + * The random program generator. + * + * Generates a single well-typed function in the engine's subset (`docs/semantics.md` section 7), + * biased toward the shapes that broke the checker before: loops with `break`/`continue` in every + * position, accumulators carried across iterations, nested loops, a `switch` inside a loop, + * indices derived from a loop counter, arithmetic that walks toward a range's bounds, string + * building, and early `return` from inside a loop. + * + * Every mutable integer local is declared `Int` (the full platform-safe domain), never a tight + * `IntRange`. That is not a simplification that avoids the interesting bugs — the checker still + * infers and proves a *tight* range for the local on every assignment (that tight range is what a + * function's declared return type would need to subtype, and it is what `values.ts`'s + * `withinType` checks the interpreter's real answer against); a wide declared ceiling only means + * an accumulator never gets rejected for growing wider than its author's intent, which is not + * the analysis this generator is trying to exercise. This is what makes free-form generation + * tractable without re-implementing the checker's interval arithmetic to predict the exact range + * an annotation would need up front. + * + * A `seq.get(list, index)` (or `list[index]`) is generated only where the generator itself can + * see the index is safe — a counted loop's own counter, ranging over a *fixed*-length list's + * `.length` — which mirrors the loop shape the `break`/`continue` fixpoint bug actually lived in. + */ + +import { Rng } from "./rng.ts"; +import type { BinOp, Expr, FuzzFunc, Param, Stmt } from "./ast.ts"; + +type IntInfo = { readonly kind: "int"; readonly lo: bigint; readonly hi: bigint }; +type ListInfo = { + readonly kind: "list"; + readonly elemLo: bigint; + readonly elemHi: bigint; + readonly min: number; + readonly max: number; +}; +type BoolInfo = { readonly kind: "bool" }; +type EnumInfo = { readonly kind: "enum" }; +type VarInfo = IntInfo | ListInfo | BoolInfo | EnumInfo; + +type AccKind = "int" | "bool" | "string"; + +class Ctx { + readonly rng: Rng; + readonly scope = new Map(); + readonly mutable = new Set(); + /** index variable name -> list variable names it is proven in range for. */ + readonly safeIndex = new Map>(); + enumMembers: readonly string[] | undefined; + accName = "acc"; + accKind: AccKind = "int"; + private counter = 0; + + constructor(rng: Rng) { + this.rng = rng; + } + + fresh(prefix: string): string { + this.counter += 1; + return `${prefix}${this.counter}`; + } + + intVars(exclude: readonly string[] = []): string[] { + return [...this.scope.entries()] + .filter(([name, info]) => info.kind === "int" && !exclude.includes(name)) + .map(([name]) => name); + } + + boolVars(exclude: readonly string[] = []): string[] { + return [...this.scope.entries()] + .filter(([name, info]) => info.kind === "bool" && !exclude.includes(name)) + .map(([name]) => name); + } + + fixedListVars(): string[] { + return [...this.scope.entries()] + .filter(([, info]) => info.kind === "list" && info.min === info.max) + .map(([name]) => name); + } + + listVars(): string[] { + return [...this.scope.entries()].filter(([, info]) => info.kind === "list").map(([name]) => name); + } +} + +/** A leaf integer: a small literal, or an in-scope integer variable (not the accumulator). */ +function genIntLeaf(ctx: Ctx): Expr { + const candidates = ctx.intVars([ctx.accName]); + const options: (readonly [number, () => Expr])[] = [ + [3, () => ({ k: "int", value: BigInt(ctx.rng.int(-20, 20)) })], + ]; + if (candidates.length > 0) { + options.push([4, () => ({ k: "var", name: ctx.rng.pick(candidates) })]); + } + const fixedLists = ctx.fixedListVars(); + // An element access through a counter proven safe for that list — the pattern the historical + // `break`/`continue` bug lived in, so it is weighted heavily. + const safePairs: (readonly [string, string])[] = []; + for (const [indexVar, lists] of ctx.safeIndex) { + for (const list of lists) if (fixedLists.includes(list)) safePairs.push([indexVar, list]); + } + if (safePairs.length > 0) { + options.push([ + 5, + () => { + const [indexVar, list] = ctx.rng.pick(safePairs); + return { k: "index", target: { k: "var", name: list }, index: { k: "var", name: indexVar } }; + }, + ]); + } + return ctx.rng.weighted(options)(); +} + +function genIntExpr(ctx: Ctx, depth: number): Expr { + if (depth <= 0 || ctx.rng.bool(0.45)) return genIntLeaf(ctx); + const op = ctx.rng.pick(["+", "-", "*", "/", "%"]); + if (op === "/" || op === "%") { + // The divisor must be provably non-zero; a positive literal is always provable, and using + // one on both sides of the idiom-mode boundary is exactly what exercises the Python + // truncated-vs-floored divergence documented in docs/semantics.md section 2.1. + const divisor: Expr = { k: "int", value: BigInt(ctx.rng.int(1, 11)) }; + return { k: "bin", op, l: genIntExpr(ctx, depth - 1), r: divisor }; + } + return { k: "bin", op, l: genIntExpr(ctx, depth - 1), r: genIntExpr(ctx, depth - 1) }; +} + +function genBoolExpr(ctx: Ctx, depth: number): Expr { + const boolVars = ctx.boolVars(); + if (depth <= 0 || ctx.rng.bool(0.3)) { + const options: (readonly [number, () => Expr])[] = [ + [2, () => ({ k: "bool", value: ctx.rng.bool() })], + [ + 5, + () => ({ + k: "bin", + op: ctx.rng.pick(["<", "<=", ">", ">=", "===", "!=="]), + l: genIntExpr(ctx, 1), + r: genIntExpr(ctx, 1), + }), + ], + ]; + if (boolVars.length > 0) options.push([2, () => ({ k: "var", name: ctx.rng.pick(boolVars) })]); + return ctx.rng.weighted(options)(); + } + if (ctx.rng.bool(0.2)) return { k: "not", e: genBoolExpr(ctx, depth - 1) }; + return { + k: "logical", + op: ctx.rng.pick(["&&", "||"]), + l: genBoolExpr(ctx, depth - 1), + r: genBoolExpr(ctx, depth - 1), + }; +} + +/** An ASCII printable code point (32..126), for the string-building accumulator. */ +function genCodePointExpr(ctx: Ctx): Expr { + const raw = genIntLeaf(ctx); + // Fold whatever came out into the printable range with a portable, always-safe idiom, rather + // than trying to prove the leaf itself lands there — a `% 95 + 32` is provably in range from + // `rangeMod`'s own formula, no matter what the dividend's range is. + const shifted: Expr = { k: "bin", op: "+", l: raw, r: { k: "int", value: 10_000n } }; + const folded: Expr = { k: "bin", op: "%", l: shifted, r: { k: "int", value: 95n } }; + return { k: "bin", op: "+", l: folded, r: { k: "int", value: 32n } }; +} + +function accUpdateStmt(ctx: Ctx): Stmt { + if (ctx.accKind === "bool") { + return { k: "assign", name: ctx.accName, op: "=", value: genBoolExpr(ctx, 2) }; + } + if (ctx.accKind === "string") { + return { k: "push", name: ctx.accName, value: genCodePointExpr(ctx) }; + } + const op = ctx.rng.pick<"+=" | "-=" | "*=">(["+=", "-=", "*="]); + // `*=` compounds fast; keep its right-hand side a small literal so the true accumulated value + // (which must still fit the generator's own sense of "small", see the module doc) does not + // explode across a loop with dozens of iterations. + const value = op === "*=" ? { k: "int" as const, value: BigInt(ctx.rng.int(-3, 3)) } : genIntExpr(ctx, 2); + return { k: "assign", name: ctx.accName, op, value }; +} + +/** Statements available inside a loop body, weighted toward the bug-relevant shapes. */ +function genLoopStmt(ctx: Ctx, loopDepth: number): Stmt { + const options: (readonly [number, () => Stmt])[] = [ + [5, () => accUpdateStmt(ctx)], + [3, () => ({ k: "if", test: genBoolExpr(ctx, 2), then: [{ k: "break" }] })], + [3, () => ({ k: "if", test: genBoolExpr(ctx, 2), then: [{ k: "continue" }] })], + [ + 3, + () => ({ + k: "if", + test: genBoolExpr(ctx, 1), + then: [accUpdateStmt(ctx), { k: "break" }], + else_: [accUpdateStmt(ctx)], + }), + ], + ]; + if (ctx.accKind !== "string") { + options.push([ + 1, + () => ({ + k: "if", + test: genBoolExpr(ctx, 1), + then: [{ k: "return", value: { k: "var", name: ctx.accName } }], + }), + ]); + } + if (ctx.enumMembers !== undefined && ctx.scope.has("kind")) { + options.push([4, () => genSwitch(ctx)]); + } + if (loopDepth > 0 && ctx.listVars().length > 0) { + options.push([3, () => genLoop(ctx, loopDepth - 1)]); + } + return ctx.rng.weighted(options)(); +} + +function genSwitch(ctx: Ctx): Stmt { + const members = ctx.enumMembers!; + // A random non-empty, non-total subset gets an explicit case; `default` always covers the + // rest, so the switch is exhaustive (E_NON_EXHAUSTIVE) no matter how many members are named. + const shuffled = [...members].sort(() => ctx.rng.float() - 0.5); + const explicitCount = ctx.rng.int(1, Math.max(1, members.length - 1)); + const cases: { readonly test?: string; readonly body: readonly Stmt[] }[] = shuffled + .slice(0, explicitCount) + .map((member) => ({ test: `"${member}"`, body: [accUpdateStmt(ctx), { k: "break" as const }] })); + cases.push({ body: [accUpdateStmt(ctx), { k: "break" as const }] }); + return { k: "switch", subject: { k: "var", name: "kind" }, cases }; +} + +/** A loop over a list: counted over a fixed-length list's `.length` (unlocking a safe index), or `for…of`. */ +function genLoop(ctx: Ctx, loopDepth: number): Stmt { + const fixed = ctx.fixedListVars(); + const useCounted = fixed.length > 0 && ctx.rng.bool(0.6); + if (useCounted) { + const list = ctx.rng.pick(fixed); + const info = ctx.scope.get(list) as ListInfo; + const name = ctx.fresh("i"); + ctx.scope.set(name, { kind: "int", lo: 0n, hi: BigInt(info.max - 1) }); + const forSameLength = new Set(fixed.filter((other) => (ctx.scope.get(other) as ListInfo).max === info.max)); + ctx.safeIndex.set(name, forSameLength); + const bodyLength = ctx.rng.int(1, 3); + const body = Array.from({ length: bodyLength }, () => genLoopStmt(ctx, loopDepth)); + ctx.safeIndex.delete(name); + ctx.scope.delete(name); + return { k: "forCounted", name, fromN: 0n, to: { k: "length", target: { k: "var", name: list } }, body }; + } + const list = ctx.rng.pick(ctx.listVars()); + const info = ctx.scope.get(list) as ListInfo; + const name = ctx.fresh("e"); + ctx.scope.set(name, { kind: "int", lo: info.elemLo, hi: info.elemHi }); + const bodyLength = ctx.rng.int(1, 3); + const body = Array.from({ length: bodyLength }, () => genLoopStmt(ctx, loopDepth)); + ctx.scope.delete(name); + return { k: "forOf", name, iterable: { k: "var", name: list }, body }; +} + +export type GenerateOptions = { + /** Nesting depth of loops inside loops (the generator never goes deeper than this). */ + readonly maxLoopDepth?: number; + /** + * A unique suffix for the function's and its enum type's names, so several generated programs + * can share one module (one compile, instead of one per program) without colliding. + */ + readonly id?: string; +}; + +export function generateProgram(seed: number, options: GenerateOptions = {}): FuzzFunc { + const rng = new Rng(seed); + const ctx = new Ctx(rng); + const maxLoopDepth = options.maxLoopDepth ?? 2; + const id = options.id ?? "0"; + const enumName = `Kind_${id}`; + + const useEnum = rng.bool(0.45); + let enumDecl: FuzzFunc["enumDecl"]; + if (useEnum) { + const memberCount = rng.int(2, 4); + const members = Array.from({ length: memberCount }, (_, i) => String.fromCharCode(65 + i)); + enumDecl = { name: enumName, members }; + ctx.enumMembers = members; + } + + const params: Param[] = []; + const paramCount = rng.int(1, 3) + (useEnum ? 1 : 0); + let enumPlaced = !useEnum; + let usedWide = false; + for (let i = 0; i < paramCount; i++) { + if (!enumPlaced && (i === paramCount - 1 || rng.bool(0.4))) { + params.push({ name: "kind", typeAnn: enumName }); + ctx.scope.set("kind", { kind: "enum" }); + enumPlaced = true; + continue; + } + const name = ctx.fresh("p"); + const kind = rng.weighted<"fixedList" | "varList" | "scalar" | "bool">([ + [4, "fixedList"], + [2, "varList"], + [3, "scalar"], + [1, "bool"], + ]); + if (kind === "bool") { + params.push({ name, typeAnn: "boolean" }); + ctx.scope.set(name, { kind: "bool" }); + } else if (kind === "scalar") { + const lo = BigInt(rng.int(-1000, 0)); + const hi = lo + BigInt(rng.int(1, 2000)); + params.push({ name, typeAnn: `IntRange<${lo}, ${hi}>` }); + ctx.scope.set(name, { kind: "int", lo, hi }); + } else { + const elemLo = BigInt(rng.int(-5, 0)); + const elemHi = elemLo + BigInt(rng.int(3, 15)); + // Occasionally push a list past the exact-fixpoint threshold (64 trip counts, see + // docs/semantics.md section 2.2) so the widen/clamp path gets exercised too, not just + // the small exact case the historical bug happened to trip on. Capped at one per + // function: the checker re-checks a loop body once per fixpoint iteration of every loop + // it is nested inside, so two independently "wide" loops nested inside one another would + // multiply the 64-iteration fixpoint against itself and make a single generated program + // expensive to typecheck for no extra coverage. + const wide = !usedWide && rng.bool(0.06); + if (wide) usedWide = true; + if (kind === "fixedList") { + const len = wide ? rng.int(65, 90) : rng.int(1, 7); + params.push({ name, typeAnn: `List, ${len}, ${len}>` }); + ctx.scope.set(name, { kind: "list", elemLo, elemHi, min: len, max: len }); + } else { + const min = rng.int(0, 2); + const max = min + (wide ? rng.int(63, 90) : rng.int(1, 6)); + params.push({ name, typeAnn: `List, ${min}, ${max}>` }); + ctx.scope.set(name, { kind: "list", elemLo, elemHi, min, max }); + } + } + } + + // At least one list-shaped parameter is guaranteed: a function with no list to iterate is + // well-typed but useless for the loop/break/continue bugs this generator hunts. + if (ctx.listVars().length === 0) { + const name = ctx.fresh("p"); + const elemLo = BigInt(rng.int(-5, 0)); + const elemHi = elemLo + BigInt(rng.int(3, 15)); + const len = rng.int(1, 7); + params.push({ name, typeAnn: `List, ${len}, ${len}>` }); + ctx.scope.set(name, { kind: "list", elemLo, elemHi, min: len, max: len }); + } + + const accKind: AccKind = rng.weighted([ + [5, "int"], + [2, "bool"], + [3, "string"], + ]); + // String building needs a list of code points to draw from; fall back to `int` otherwise. + ctx.accKind = accKind === "string" && ctx.listVars().length === 0 ? "int" : accKind; + + const body: Stmt[] = []; + if (ctx.accKind === "bool") { + body.push({ + k: "let", + name: "acc", + typeAnn: "boolean", + init: { k: "bool", value: rng.bool() }, + mutable: true, + }); + } else if (ctx.accKind === "string") { + body.push({ + k: "let", + name: "acc", + typeAnn: "List, 0, 4096>", + init: { k: "emptyArray" }, + mutable: true, + }); + } else { + body.push({ + k: "let", + name: "acc", + typeAnn: "Int", + init: { k: "int", value: BigInt(rng.int(-10, 10)) }, + mutable: true, + }); + } + // Registered only for `bool`: `genIntLeaf` explicitly excludes the accumulator by name (an + // `int` accumulator only ever appears on the left of its own `+=`/`-=`/`*=`, never inside its + // own right-hand side), and a `string` accumulator is a list, never a plain variable reference. + if (ctx.accKind === "bool") ctx.scope.set("acc", { kind: "bool" }); + + const loopCount = rng.int(1, 3); + for (let i = 0; i < loopCount; i++) { + if (ctx.listVars().length === 0) break; + // `genLoop`'s own body is generated at the depth passed in (not `- 1`: the budget is spent + // when a *nested* loop is chosen, inside `genLoopStmt`), so passing `maxLoopDepth` here would + // let a top-level loop plus two levels of nesting through — three loops deep, not two — and + // three loops each ranging over a "wide" (up to 90-element) list compounds into roughly + // 90^3 body executions, which is where the generator itself, not anything it is testing, + // becomes the bottleneck. + body.push(genLoop(ctx, maxLoopDepth - 1)); + } + + if (ctx.accKind === "string") { + body.push({ k: "return", value: { k: "call", fn: "str.fromCodePoints", args: [{ k: "var", name: "acc" }] } }); + } else { + body.push({ k: "return", value: { k: "var", name: "acc" } }); + } + + const retTypeAnn = ctx.accKind === "bool" ? "boolean" : ctx.accKind === "string" ? "Ascii" : "Int"; + + return { enumDecl, params, retTypeAnn, body, fnName: `fuzz_${id}` }; +} diff --git a/engine/src/fuzz/harness.ts b/engine/src/fuzz/harness.ts new file mode 100644 index 000000000..ebf25e130 --- /dev/null +++ b/engine/src/fuzz/harness.ts @@ -0,0 +1,333 @@ +/** + * Fast and full differential runs over generated programs. + * + * Fast mode needs no code generation at all: it compiles a generated program, runs the reference + * interpreter, and checks that every value the interpreter actually produced lies inside the + * range, length and character class the checker proved for it (`docs/semantics.md` section 10, + * Layer 1 — see the module doc on `generate.ts` for why this is the highest-value comparison + * here). Full mode adds all four generated targets, in both idiom modes, reusing the same + * `runInterpreter`/`runTarget`/`compare` machinery `core/conformance/run.ts` drives by hand. + */ + +import { mkdtempSync, mkdirSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join, resolve } from "node:path"; +import { CompileError, compileProject } from "../api.ts"; +import type { Compilation } from "../api.ts"; +import { generate, writeFiles } from "../backend/generate.ts"; +import { TYPESCRIPT_BACKEND } from "../targets/typescript/index.ts"; +import { PYTHON_BACKEND } from "../targets/python/index.ts"; +import { GO_BACKEND } from "../targets/go/index.ts"; +import { RUST_BACKEND } from "../targets/rust/index.ts"; +import { compare, runInterpreter, runTarget } from "../conformance/differential.ts"; +import type { Case, Divergence, TargetRunner } from "../conformance/differential.ts"; +import { DomainFailure } from "../intrinsics/index.ts"; +import { Interpreter } from "../interp/interp.ts"; +import type { CFunc } from "../core/ir.ts"; +import type { Value } from "../values.ts"; +import type { FuzzFunc } from "./ast.ts"; +import { printProgram } from "./ast.ts"; +import { generateProgram } from "./generate.ts"; +import { Rng, subSeed } from "./rng.ts"; +import type { Fails } from "./shrink.ts"; +import { shrinkInputs, shrinkProgram } from "./shrink.ts"; +import { randomValue, withinType } from "./values.ts"; + +const BACKENDS = { + typescript: TYPESCRIPT_BACKEND, + python: PYTHON_BACKEND, + go: GO_BACKEND, + rust: RUST_BACKEND, +} as const; + +/** A scratch project directory holding exactly one source module, overwritten on every call. */ +class Scratch { + readonly dir: string; + readonly sourceDir: string; + + constructor() { + this.dir = mkdtempSync(join(tmpdir(), "logic-engine-fuzz-")); + this.sourceDir = join(this.dir, "source"); + mkdirSync(this.sourceDir, { recursive: true }); + } + + /** Compiles `source` as `program.ts`, the scratch project's one module. */ + compile(source: string): Compilation | undefined { + writeFileSync(join(this.sourceDir, "program.ts"), source); + try { + return compileProject(this.sourceDir); + } catch (error) { + if (error instanceof CompileError) return undefined; + throw error; + } + } + + cleanup(): void { + rmSync(this.dir, { recursive: true, force: true }); + } +} + +export type BoundsViolation = { + readonly seed: number; + readonly index: number; + readonly fnName: string; + readonly input: readonly Value[]; + readonly produced: unknown; + readonly retType: string; +}; + +export type FastReport = { + readonly seed: number; + readonly attempted: number; + readonly compiled: number; + readonly casesRun: number; + readonly violations: (BoundsViolation & { readonly source: string })[]; + readonly elapsedMs: number; +}; + +function qualifiedName(fn: FuzzFunc): string { + return `program::${fn.fnName}`; +} + +/** + * Calls the reference interpreter directly, keeping its native `Value` representation (a bigint + * `Int`, a real array `List`, …) rather than the JSON `toJson`/`fromJson` `differential.ts` uses + * for the cross-target protocol — the whole point of Layer 1 is checking that native value + * against the checker's own proven type, which needs the exact representation, not JSON's. + */ +function interpretRaw( + compilation: Compilation, + name: string, + args: readonly Value[], +): { ok: true; value: Value } | { ok: false; error: string } { + try { + return { ok: true, value: new Interpreter(compilation.program).call(name, args) }; + } catch (error) { + if (error instanceof DomainFailure) return { ok: false, error: error.errorType }; + throw error; + } +} + +export type ProgressCb = (attempted: number, compiled: number) => void; + +export function runFast( + seed: number, + count: number, + casesPerProgram = 8, + onProgress?: ProgressCb, +): FastReport { + const start = Date.now(); + const scratch = new Scratch(); + let compiled = 0; + let casesRun = 0; + const violations: (BoundsViolation & { source: string })[] = []; + try { + for (let i = 0; i < count; i++) { + const programSeed = subSeed(seed, i); + const fn = generateProgram(programSeed, { id: String(i) }); + const source = printProgram(fn); + const compilation = scratch.compile(source); + if (compilation === undefined) { + onProgress?.(i + 1, compiled); + continue; + } + compiled += 1; + const cfunc = compilation.program.functions.get(qualifiedName(fn)); + if (cfunc === undefined) { + onProgress?.(i + 1, compiled); + continue; + } + const rng = new Rng(subSeed(programSeed, 0xf057)); + for (let c = 0; c < casesPerProgram; c++) { + const input = cfunc.params.map((p) => randomValue(rng, p.type)); + casesRun += 1; + const outcome = interpretRaw(compilation, cfunc.name, input); + if (outcome.ok && !withinType(outcome.value, cfunc.ret)) { + violations.push({ + seed, + index: i, + fnName: fn.fnName, + input, + produced: outcome.value, + retType: cfunc.ret.kind, + source, + }); + break; + } + } + onProgress?.(i + 1, compiled); + } + } finally { + scratch.cleanup(); + } + return { seed, attempted: count, compiled, casesRun, violations, elapsedMs: Date.now() - start }; +} + +/** Re-shrinks a bounds violation found by `runFast` into the smallest reproducing program. */ +export function shrinkViolation( + violation: BoundsViolation & { source: string }, + casesPerProgram: number, +): { fn: FuzzFunc; source: string; input: readonly Value[] } { + const scratch = new Scratch(); + try { + const original = generateProgram(subSeed(violation.seed, violation.index), { id: String(violation.index) }); + const fails: Fails = (candidate) => { + const compilation = scratch.compile(printProgram(candidate)); + if (compilation === undefined) return { ok: false }; + const cfunc = compilation.program.functions.get(qualifiedName(candidate)); + if (cfunc === undefined) return { ok: false }; + // The original failing input is tried first: most statement removals do not touch the + // path that produced it, so it usually still reproduces without any extra search. + for (const input of [violation.input, ...extraInputs(cfunc, violation.seed, casesPerProgram)]) { + const outcome = interpretRaw(compilation, cfunc.name, input); + if (outcome.ok && !withinType(outcome.value, cfunc.ret)) return { ok: true, input }; + } + return { ok: false }; + }; + const { fn: shrunkFn, input } = shrinkProgram(original, fails, 500); + const finalInput = input ?? violation.input; + const compilation = scratch.compile(printProgram(shrunkFn))!; + const cfunc = compilation.program.functions.get(qualifiedName(shrunkFn))!; + const narrowedInput = shrinkInputs( + finalInput, + cfunc.params.map((p) => p.type), + (candidate) => { + const outcome = interpretRaw(compilation, cfunc.name, candidate); + return outcome.ok && !withinType(outcome.value, cfunc.ret); + }, + ); + return { fn: shrunkFn, source: printProgram(shrunkFn), input: narrowedInput }; + } finally { + scratch.cleanup(); + } +} + +function extraInputs(cfunc: CFunc, seed: number, count: number): readonly (readonly Value[])[] { + const rng = new Rng(subSeed(seed, 0xbeef)); + return Array.from({ length: count }, () => cfunc.params.map((p) => randomValue(rng, p.type))); +} + +/* -------------------------------------------------------------------------------------------- * + * Full mode: interpreter and all four targets, in both idiom modes. + * -------------------------------------------------------------------------------------------- */ + +export type FullReport = { + readonly seed: number; + readonly attempted: number; + readonly compiled: number; + readonly divergences: (Divergence & { readonly source: string })[]; + readonly elapsedMs: number; +}; + +function runners( + outRoot: string, + suffix: string, + targets: readonly (keyof typeof BACKENDS)[], +): readonly TargetRunner[] { + const all: Record = { + typescript: { + name: `typescript${suffix}`, + command: process.execPath, + args: ["_driver.ts"], + cwd: resolve(outRoot, `typescript${suffix}`), + }, + python: { name: `python${suffix}`, command: "python3", args: ["-m", `python${suffix}._driver`], cwd: outRoot }, + go: { name: `go${suffix}`, command: "go", args: ["run", "./cmd/driver"], cwd: resolve(outRoot, `go${suffix}`) }, + rust: { + name: `rust${suffix}`, + command: "cargo", + args: ["run", "--offline", "--release", "--quiet", "--bin", "driver"], + cwd: resolve(outRoot, `rust${suffix}`), + }, + }; + return targets.map((target) => all[target]); +} + +export function runFull( + seed: number, + count: number, + casesPerProgram = 4, + targets: readonly (keyof typeof BACKENDS)[] = ["typescript", "python", "go", "rust"], + onProgress?: ProgressCb, +): FullReport { + const start = Date.now(); + const prefilter = new Scratch(); + const survivors: { index: number; fn: FuzzFunc; source: string }[] = []; + try { + for (let i = 0; i < count; i++) { + const programSeed = subSeed(seed, i); + const fn = generateProgram(programSeed, { id: String(i) }); + const source = printProgram(fn); + if (prefilter.compile(source) !== undefined) survivors.push({ index: i, fn, source }); + onProgress?.(i + 1, survivors.length); + } + } finally { + prefilter.cleanup(); + } + + const batchSource = survivors.map((s) => s.source).join("\n"); + const project = mkdtempSync(join(tmpdir(), "logic-engine-fuzz-full-")); + const sourceDir = join(project, "source"); + const outRoot = join(project, "out"); + mkdirSync(sourceDir, { recursive: true }); + writeFileSync(join(sourceDir, "program.ts"), batchSource); + + const divergences: (Divergence & { source: string })[] = []; + try { + const compilation = compileProject(sourceDir); + const rng = new Rng(subSeed(seed, 0xfeed)); + const cases: Case[] = []; + const caseSource = new Map(); + for (const { fn, source } of survivors) { + const cfunc = compilation.program.functions.get(qualifiedName(fn)); + if (cfunc === undefined) continue; + caseSource.set(cfunc.name, source); + for (let c = 0; c < casesPerProgram; c++) { + cases.push({ fn: cfunc.name, args: cfunc.params.map((p) => randomValue(rng, p.type)) }); + } + } + if (cases.length === 0) { + return { seed, attempted: count, compiled: survivors.length, divergences: [], elapsedMs: Date.now() - start }; + } + const reference = runInterpreter(compilation.program, cases); + + for (const mode of ["idiomatic", "plain"] as const) { + const suffix = mode === "plain" ? "-plain" : ""; + for (const target of targets) { + const backend = BACKENDS[target]; + const result = generate(compilation.program, backend, { noIdioms: mode === "plain" }); + const outDir = resolve(outRoot, `${target}${suffix}`); + writeFiles(outDir, result, { + "LOWERING.md": result.lowering, + "API.json": `${JSON.stringify(result.api, null, "\t")}\n`, + "SOURCEMAP.json": `${JSON.stringify(result.sourceMap, null, "\t")}\n`, + }); + } + for (const runner of runners(outRoot, suffix, targets)) { + let actual; + try { + actual = runTarget(runner, cases); + } catch (error) { + // A target that fails to build at all is a finding in its own right, reported as a + // single divergence rather than silently skipped. + divergences.push({ + target: runner.name, + case: cases[0]!, + expected: reference[0]!, + actual: { ok: false, error: `driver failed to run: ${String(error)}` }, + source: "(build/run failure, not a single program)", + } as Divergence & { source: string }); + continue; + } + const found = compare(reference, actual, cases, runner.name); + for (const divergence of found) { + divergences.push({ ...divergence, source: caseSource.get(divergence.case.fn) ?? "(unknown)" }); + } + } + } + } finally { + rmSync(project, { recursive: true, force: true }); + } + + return { seed, attempted: count, compiled: survivors.length, divergences, elapsedMs: Date.now() - start }; +} diff --git a/engine/src/fuzz/rng.ts b/engine/src/fuzz/rng.ts new file mode 100644 index 000000000..8576b6ab0 --- /dev/null +++ b/engine/src/fuzz/rng.ts @@ -0,0 +1,94 @@ +/** + * A hand-rolled seeded PRNG (splitmix32 stream, mulberry32 mixing). + * + * The generator never takes a dependency on a general-purpose property-testing library (see + * `docs/fuzzing.md` for why): a 32-bit generator with a handful of derived helpers is all a + * program generator needs, and it keeps `engine/`'s dependency count exactly where it was. + * + * Every method is a pure function of the current state, so replaying a seed reproduces the exact + * same sequence of decisions on any machine, forever — the whole point of printing a seed on + * failure. + */ + +export class Rng { + private state: number; + + constructor(seed: number) { + // Fold the seed through a couple of rounds before use so that nearby seeds (1, 2, 3, …, as a + // human picks when replaying) do not produce visibly correlated first draws. + this.state = (seed ^ 0x9e3779b9) >>> 0; + this.next(); + this.next(); + } + + /** Uniform in `[0, 2^32)`. */ + private next(): number { + // mulberry32 + this.state = (this.state + 0x6d2b79f5) >>> 0; + let t = this.state; + t = Math.imul(t ^ (t >>> 15), t | 1); + t ^= t + Math.imul(t ^ (t >>> 7), t | 61); + return ((t ^ (t >>> 14)) >>> 0) / 4294967296; + } + + /** A float in `[0, 1)`. */ + float(): number { + return this.next(); + } + + /** True with probability `p` (default: a fair coin). */ + bool(p = 0.5): boolean { + return this.next() < p; + } + + /** An integer in `[lo, hi]` inclusive, as a plain number. Both bounds are safe integers. */ + int(lo: number, hi: number): number { + if (hi < lo) throw new Error(`Rng.int: empty range [${lo}, ${hi}]`); + return lo + Math.floor(this.next() * (hi - lo + 1)); + } + + /** An integer in `[lo, hi]` inclusive, as a bigint, for ranges too wide for `number`. */ + bigint(lo: bigint, hi: bigint): bigint { + if (hi < lo) throw new Error(`Rng.bigint: empty range [${lo}, ${hi}]`); + const span = hi - lo + 1n; + if (span <= 0x100000000n) { + return lo + BigInt(this.int(0, Number(span) - 1)); + } + // A wide span (only ever the platform-safe domain in this generator): compose two 32-bit + // draws rather than lose precision to `float() * Number(span)`. + const hiPart = BigInt(this.int(0, 0xffffffff)); + const loPart = BigInt(this.int(0, 0xffffffff)); + return lo + ((hiPart << 32n) | loPart) % span; + } + + /** One element of a non-empty array. */ + pick(items: readonly T[]): T { + if (items.length === 0) throw new Error("Rng.pick: empty array"); + return items[this.int(0, items.length - 1)]!; + } + + /** An index into a weighted list of options, each `[weight, value]`. */ + weighted(options: readonly (readonly [number, T])[]): T { + const total = options.reduce((sum, [w]) => sum + w, 0); + let draw = this.float() * total; + for (const [weight, value] of options) { + draw -= weight; + if (draw <= 0) return value; + } + return options[options.length - 1]![1]; + } +} + +/** + * Derives an independent-looking sub-seed for program `index` under a master `seed`, so a batch + * of N generated programs is addressable one at a time: replaying `(seed, index)` reproduces + * exactly the program the batch produced at that position, independent of batch size or of what + * happened to any other program in the run. + */ +export function subSeed(seed: number, index: number): number { + // splitmix32 step: cheap, well-mixed, and needs no state beyond the two inputs. + let z = (seed + Math.imul(index + 1, 0x9e3779b9)) >>> 0; + z = Math.imul(z ^ (z >>> 16), 0x85ebca6b) >>> 0; + z = Math.imul(z ^ (z >>> 13), 0xc2b2ae35) >>> 0; + return (z ^ (z >>> 16)) >>> 0; +} diff --git a/engine/src/fuzz/shrink.ts b/engine/src/fuzz/shrink.ts new file mode 100644 index 000000000..a0f7809a9 --- /dev/null +++ b/engine/src/fuzz/shrink.ts @@ -0,0 +1,172 @@ +/** + * Shrinking: turns a failing generated program into the smallest one that still fails the same + * way, so a report never hands back the raw output of a random search. + * + * Two independent axes, both delta-debugging (try a smaller candidate; keep it only if the + * failure still reproduces; repeat to a fixpoint): + * + * - `shrinkInputs` narrows the failing argument tuple toward the edge of each parameter's own + * type (zero, an empty or shorter list, `false`) without touching the program at all. + * - `shrinkProgram` narrows the program itself: dropping a statement, an `if`'s `else`, or a + * `switch` case, recompiling after every trial so a candidate that no longer type-checks is + * never accepted. + */ + +import type { SemType } from "../types.ts"; +import type { Value } from "../values.ts"; +import type { FuzzFunc, Stmt } from "./ast.ts"; + +export type Fails = ( + fn: FuzzFunc, +) => { readonly ok: true; readonly input?: readonly Value[] } | { readonly ok: false }; + +/** Every statement-list-valued position reachable one step into `stmt`, paired with a rebuilder. */ +function* statementVariants(stmt: Stmt): Generator { + switch (stmt.k) { + case "if": + if (stmt.else_ !== undefined) yield { ...stmt, else_: undefined }; + for (const body of blockVariants(stmt.then)) yield { ...stmt, then: body }; + if (stmt.else_ !== undefined) { + for (const body of blockVariants(stmt.else_)) yield { ...stmt, else_: body }; + } + return; + case "forCounted": + for (const body of blockVariants(stmt.body)) yield { ...stmt, body }; + return; + case "forOf": + for (const body of blockVariants(stmt.body)) yield { ...stmt, body }; + return; + case "switch": { + for (let i = 0; i < stmt.cases.length; i++) { + if (stmt.cases[i]!.test === undefined) continue; + yield { ...stmt, cases: [...stmt.cases.slice(0, i), ...stmt.cases.slice(i + 1)] }; + } + for (let i = 0; i < stmt.cases.length; i++) { + for (const body of blockVariants(stmt.cases[i]!.body)) { + const cases = [...stmt.cases]; + cases[i] = { ...cases[i]!, body }; + yield { ...stmt, cases }; + } + } + return; + } + default: + return; + } +} + +/** Every smaller variant of a statement list: one statement dropped, or one nested shrink applied. */ +function* blockVariants(stmts: readonly Stmt[]): Generator { + for (let i = 0; i < stmts.length; i++) { + yield [...stmts.slice(0, i), ...stmts.slice(i + 1)]; + } + for (let i = 0; i < stmts.length; i++) { + for (const replacement of statementVariants(stmts[i]!)) { + yield [...stmts.slice(0, i), replacement, ...stmts.slice(i + 1)]; + } + } +} + +/** + * Repeatedly replaces the program with the first smaller variant that still fails, until a full + * pass over the current program finds none. `budget` bounds how many candidates are tried in + * total, so a pathological program cannot make a report hang. + */ +export function shrinkProgram( + fn: FuzzFunc, + fails: Fails, + budget = 400, +): { fn: FuzzFunc; input?: readonly Value[] } { + let current = fn; + let bestInput: readonly Value[] | undefined; + let spent = 0; + let improved = true; + while (improved && spent < budget) { + improved = false; + for (const body of blockVariants(current.body)) { + if (spent >= budget) break; + spent += 1; + const candidate: FuzzFunc = { ...current, body }; + const result = fails(candidate); + if (!result.ok) continue; + current = candidate; + if (result.input !== undefined) bestInput = result.input; + improved = true; + break; + } + } + return { fn: current, input: bestInput }; +} + +/** Every literal integer reachable inside an expression, for shrinking a value toward zero. */ +function shrinkIntTarget(lo: bigint, hi: bigint, value: bigint): bigint | undefined { + if (value === 0n) return undefined; + const towardZero = value > 0n ? value - 1n : value + 1n; + // Halving converges in O(log n) instead of O(n) for a wide range; either way the candidate + // must stay inside the type's own proven bounds. + const halved = value / 2n; + const candidate = halved !== value ? halved : towardZero; + return candidate < lo ? lo : candidate > hi ? hi : candidate; +} + +/** + * Narrows a failing input tuple toward the smallest values, in place, at each parameter's own + * type — no recompilation involved, since the program and its declared types never change. + */ +export function shrinkInputs( + input: readonly Value[], + paramTypes: readonly SemType[], + fails: (candidate: readonly Value[]) => boolean, + budget = 300, +): readonly Value[] { + let current = [...input]; + let spent = 0; + let improved = true; + while (improved && spent < budget) { + improved = false; + for (let i = 0; i < current.length; i++) { + if (spent >= budget) break; + for (const candidateValue of shrinkOneValue(current[i]!, paramTypes[i]!)) { + if (spent >= budget) break; + spent += 1; + const trial = [...current]; + trial[i] = candidateValue; + if (fails(trial)) { + current = trial; + improved = true; + break; + } + } + if (improved) break; + } + } + return current; +} + +function* shrinkOneValue(value: Value, type: SemType): Generator { + if (type.kind === "Int" && typeof value === "bigint") { + const next = shrinkIntTarget(type.lo, type.hi, value); + if (next !== undefined) yield next; + return; + } + if (type.kind === "Bool" && typeof value === "boolean") { + if (value) yield false; + return; + } + if (type.kind === "List" && Array.isArray(value)) { + if (value.length > type.min) { + yield value.slice(0, value.length - 1); + yield value.slice(1); + } + for (let i = 0; i < value.length; i++) { + for (const smaller of shrinkOneValue(value[i]!, type.elem)) { + yield [...value.slice(0, i), smaller, ...value.slice(i + 1)]; + } + } + return; + } + if (type.kind === "String" && typeof value === "string") { + const points = [...value]; + if (points.length > type.min) yield points.slice(0, -1).join(""); + } +} diff --git a/engine/src/fuzz/values.ts b/engine/src/fuzz/values.ts new file mode 100644 index 000000000..af9afcd7c --- /dev/null +++ b/engine/src/fuzz/values.ts @@ -0,0 +1,89 @@ +/** + * Random `Value`s for a `SemType`, and the check that a `Value` really lies inside the range, + * length and character class a `SemType` claims — the "checker against reality" comparison + * (`docs/semantics.md` section 10, Layer 1), applied to whatever the generator produced rather + * than to a fixed vector table. + */ + +import type { SemType } from "../types.ts"; +import type { Value } from "../values.ts"; +import { NONE, some } from "../values.ts"; +import { Rng } from "./rng.ts"; + +/** A value drawn uniformly-ish from the type's own proven domain, biased toward its edges. */ +export function randomValue(rng: Rng, type: SemType): Value { + switch (type.kind) { + case "Bool": + return rng.bool(); + case "Int": { + // Boundary values catch off-by-one errors far more often than the middle of a range does. + const edge = rng.weighted([ + [2, type.lo], + [2, type.hi], + [1, 0n >= type.lo && 0n <= type.hi ? 0n : undefined], + [5, undefined], + ]); + return edge ?? rng.bigint(type.lo, type.hi); + } + case "String": { + const length = rng.int(type.min, Math.max(type.min, Math.min(type.max, 20))); + const points: number[] = []; + for (let i = 0; i < length; i++) { + points.push( + type.cls === "digits" + ? rng.int(0x30, 0x39) + : type.cls === "ascii" + ? rng.int(0x20, 0x7e) + : rng.int(0x20, 0x2fff), + ); + } + return String.fromCodePoint(...points); + } + case "List": { + const length = rng.int(type.min, Math.max(type.min, Math.min(type.max, 12))); + return Array.from({ length }, () => randomValue(rng, type.elem)); + } + case "Option": + return rng.bool(0.3) ? NONE : some(randomValue(rng, type.inner)); + case "Enum": + return rng.pick(type.members); + default: + throw new Error(`randomValue: unsupported type kind ${type.kind}`); + } +} + +/** Whether `value` lies inside every bound `type` claims to have proven. */ +export function withinType(value: Value, type: SemType): boolean { + switch (type.kind) { + case "Bool": + return typeof value === "boolean"; + case "Int": + return typeof value === "bigint" && value >= type.lo && value <= type.hi; + case "String": { + if (typeof value !== "string") return false; + const points = [...value].map((ch) => ch.codePointAt(0)!); + if (points.length < type.min || points.length > type.max) return false; + if (type.cls === "digits") return points.every((p) => p >= 0x30 && p <= 0x39); + if (type.cls === "ascii") return points.every((p) => p <= 0x7f); + return true; + } + case "List": { + if (!Array.isArray(value)) return false; + if (value.length < type.min || value.length > type.max) return false; + return value.every((item) => withinType(item, type.elem)); + } + case "Option": { + if (typeof value === "object" && value !== null && "__kind" in value) { + if (value.__kind === "none") return true; + if (value.__kind === "some") return withinType((value as { value: Value }).value, type.inner); + } + return false; + } + case "Enum": + return typeof value === "string" && type.members.includes(value); + default: + // Bool/Void/Never/Record/Decimal/CivilDate/… are not produced by anything this generator + // emits; reaching here would itself be a finding worth investigating by hand. + return true; + } +} diff --git a/engine/src/hir/ast.ts b/engine/src/hir/ast.ts new file mode 100644 index 000000000..e48b48aaa --- /dev/null +++ b/engine/src/hir/ast.ts @@ -0,0 +1,362 @@ +/** + * Semantic HIR: the frontend contract. + * + * The HIR is faithful to the source — sugar is still present, names are resolved, every node has + * a span — but it no longer mentions TypeScript or `oxc`. Any future frontend (a DSL, or Rust) + * produces this and nothing downstream changes. `tests/boundary.spec.ts` enforces that no module + * after this one imports the parser. + */ + +import type { Span } from "../diagnostics.ts"; + +export type HTypeExpr = + | { readonly kind: "ref"; readonly name: string; readonly args: readonly HTypeExpr[]; readonly span: Span } + | { readonly kind: "array"; readonly elem: HTypeExpr; readonly span: Span } + | { readonly kind: "union"; readonly options: readonly HTypeExpr[]; readonly span: Span } + | { readonly kind: "literal"; readonly value: string; readonly span: Span } + | { readonly kind: "object"; readonly fields: readonly HField[]; readonly span: Span } + | { + readonly kind: "func"; + readonly params: readonly HTypeExpr[]; + readonly ret: HTypeExpr; + readonly span: Span; + } + | { readonly kind: "undefined"; readonly span: Span }; + +export type HField = { + readonly name: string; + readonly type: HTypeExpr; + readonly optional: boolean; + readonly doc?: string; + readonly span: Span; +}; + +export type HTypeDecl = { + readonly name: string; + readonly type: HTypeExpr; + readonly exported: boolean; + readonly doc?: string; + readonly span: Span; +}; + +export type HErrorDecl = { + readonly name: string; + readonly base: string; + readonly exported: boolean; + readonly doc?: string; + readonly span: Span; +}; + +export type HConst = { + readonly name: string; + readonly declared?: HTypeExpr; + readonly value: HExpr; + readonly exported: boolean; + readonly span: Span; +}; + +export type HParam = { + readonly name: string; + readonly type: HTypeExpr; + readonly span: Span; +}; + +export type HFunc = { + readonly name: string; + readonly params: readonly HParam[]; + readonly ret: HTypeExpr; + readonly body: readonly HStmt[]; + readonly exported: boolean; + readonly doc?: string; + readonly span: Span; +}; + +export type HModule = { + /** Module path relative to the project's source root, without extension. */ + readonly path: string; + readonly file: string; + readonly source: string; + readonly imports: readonly HImport[]; + readonly types: readonly HTypeDecl[]; + readonly errors: readonly HErrorDecl[]; + readonly consts: readonly HConst[]; + readonly functions: readonly HFunc[]; +}; + +export type HImport = { + readonly from: string; + readonly names: readonly { readonly imported: string; readonly local: string }[]; + readonly span: Span; +}; + +export type HBinaryOp = + | "+" + | "-" + | "*" + | "/" + | "%" + | "<" + | "<=" + | ">" + | ">=" + | "===" + | "!=="; + +export type HExpr = + | { readonly kind: "int"; readonly value: bigint; readonly span: Span } + | { readonly kind: "float"; readonly value: number; readonly span: Span } + | { readonly kind: "string"; readonly value: string; readonly span: Span } + | { readonly kind: "bool"; readonly value: boolean; readonly span: Span } + | { readonly kind: "undefined"; readonly span: Span } + | { readonly kind: "regex"; readonly source: string; readonly flags: string; readonly span: Span } + | { readonly kind: "name"; readonly name: string; readonly span: Span } + | { readonly kind: "member"; readonly target: HExpr; readonly name: string; readonly span: Span } + | { readonly kind: "index"; readonly target: HExpr; readonly index: HExpr; readonly span: Span } + | { readonly kind: "call"; readonly callee: HExpr; readonly args: readonly HExpr[]; readonly span: Span } + | { readonly kind: "new"; readonly className: string; readonly args: readonly HExpr[]; readonly span: Span } + | { + readonly kind: "binary"; + readonly op: HBinaryOp; + readonly left: HExpr; + readonly right: HExpr; + readonly span: Span; + } + | { + readonly kind: "logical"; + readonly op: "&&" | "||" | "??"; + readonly left: HExpr; + readonly right: HExpr; + readonly span: Span; + } + | { readonly kind: "unary"; readonly op: "!" | "-"; readonly operand: HExpr; readonly span: Span } + | { + readonly kind: "ternary"; + readonly test: HExpr; + readonly then: HExpr; + readonly otherwise: HExpr; + readonly span: Span; + } + | { + readonly kind: "template"; + readonly parts: readonly ( + | { readonly kind: "text"; readonly value: string } + | { readonly kind: "expr"; readonly expr: HExpr } + )[]; + readonly span: Span; + } + | { + readonly kind: "object"; + readonly fields: readonly { readonly name: string; readonly value: HExpr; readonly span: Span }[]; + readonly span: Span; + } + | { readonly kind: "array"; readonly items: readonly HExpr[]; readonly span: Span } + | { + readonly kind: "lambda"; + readonly params: readonly { readonly name: string; readonly type?: HTypeExpr; readonly span: Span }[]; + readonly body: readonly HStmt[]; + readonly span: Span; + }; + +export type HStmt = + | { + readonly kind: "let"; + readonly name: string; + readonly mutable: boolean; + readonly declared?: HTypeExpr; + readonly init: HExpr; + readonly span: Span; + } + | { + readonly kind: "assign"; + readonly target: HExpr; + readonly op: "=" | "+=" | "-=" | "*="; + readonly value: HExpr; + readonly span: Span; + } + | { + readonly kind: "if"; + readonly test: HExpr; + readonly then: readonly HStmt[]; + readonly otherwise?: readonly HStmt[]; + readonly span: Span; + } + | { + readonly kind: "switch"; + readonly subject: HExpr; + readonly cases: readonly { + readonly test?: HExpr; + readonly body: readonly HStmt[]; + readonly span: Span; + }[]; + readonly span: Span; + } + | { + readonly kind: "forCounted"; + readonly name: string; + readonly from: HExpr; + readonly to: HExpr; + readonly inclusive: boolean; + readonly step: bigint; + readonly body: readonly HStmt[]; + readonly span: Span; + } + | { + readonly kind: "forOf"; + readonly name: string; + readonly iterable: HExpr; + readonly body: readonly HStmt[]; + readonly span: Span; + } + | { readonly kind: "return"; readonly value?: HExpr; readonly span: Span } + | { + readonly kind: "throw"; + readonly errorClass: string; + readonly args: readonly HExpr[]; + readonly span: Span; + } + | { readonly kind: "break"; readonly span: Span } + | { readonly kind: "continue"; readonly span: Span } + | { readonly kind: "expr"; readonly expr: HExpr; readonly span: Span } + | { readonly kind: "block"; readonly body: readonly HStmt[]; readonly span: Span }; + +/** A readable dump of the HIR, used by the stage snapshot tests. */ +export function dumpHir(module: HModule): string { + const lines: string[] = [`module ${module.path}`]; + for (const item of module.imports) { + lines.push(` import ${item.names.map((name) => name.imported).join(", ")} from ${item.from}`); + } + for (const decl of module.types) lines.push(` type ${decl.name} = ${typeExprToString(decl.type)}`); + for (const decl of module.errors) lines.push(` error ${decl.name} extends ${decl.base}`); + for (const decl of module.consts) lines.push(` const ${decl.name} = ${exprToString(decl.value)}`); + for (const decl of module.functions) { + lines.push( + ` fn ${decl.name}(${decl.params.map((param) => `${param.name}: ${typeExprToString(param.type)}`).join(", ")}): ${typeExprToString(decl.ret)}`, + ); + for (const statement of decl.body) lines.push(...stmtToString(statement, 2)); + } + return lines.join("\n"); +} + +export function typeExprToString(type: HTypeExpr): string { + switch (type.kind) { + case "ref": + return type.args.length === 0 + ? type.name + : `${type.name}<${type.args.map(typeExprToString).join(", ")}>`; + case "array": + return `${typeExprToString(type.elem)}[]`; + case "union": + return type.options.map(typeExprToString).join(" | "); + case "literal": + return JSON.stringify(type.value); + case "object": + return `{ ${type.fields.map((field) => `${field.name}${field.optional ? "?" : ""}: ${typeExprToString(field.type)}`).join("; ")} }`; + case "func": + return `(${type.params.map(typeExprToString).join(", ")}) => ${typeExprToString(type.ret)}`; + case "undefined": + return "undefined"; + default: { + const exhaustive: never = type; + return exhaustive; + } + } +} + +function exprToString(expr: HExpr): string { + switch (expr.kind) { + case "int": + return expr.value.toString(); + case "float": + return expr.value.toString(); + case "string": + return JSON.stringify(expr.value); + case "bool": + return String(expr.value); + case "undefined": + return "undefined"; + case "regex": + return `/${expr.source}/${expr.flags}`; + case "name": + return expr.name; + case "member": + return `${exprToString(expr.target)}.${expr.name}`; + case "index": + return `${exprToString(expr.target)}[${exprToString(expr.index)}]`; + case "call": + return `${exprToString(expr.callee)}(${expr.args.map(exprToString).join(", ")})`; + case "new": + return `new ${expr.className}(${expr.args.map(exprToString).join(", ")})`; + case "binary": + return `(${exprToString(expr.left)} ${expr.op} ${exprToString(expr.right)})`; + case "logical": + return `(${exprToString(expr.left)} ${expr.op} ${exprToString(expr.right)})`; + case "unary": + return `${expr.op}${exprToString(expr.operand)}`; + case "ternary": + return `(${exprToString(expr.test)} ? ${exprToString(expr.then)} : ${exprToString(expr.otherwise)})`; + case "template": + return `\`${expr.parts.map((part) => (part.kind === "text" ? part.value : `\${${exprToString(part.expr)}}`)).join("")}\``; + case "object": + return `{ ${expr.fields.map((field) => `${field.name}: ${exprToString(field.value)}`).join(", ")} }`; + case "array": + return `[${expr.items.map(exprToString).join(", ")}]`; + case "lambda": + return `(${expr.params.map((param) => param.name).join(", ")}) => …`; + default: { + const exhaustive: never = expr; + return exhaustive; + } + } +} + +function stmtToString(statement: HStmt, depth: number): string[] { + const pad = " ".repeat(depth); + switch (statement.kind) { + case "let": + return [`${pad}${statement.mutable ? "let" : "const"} ${statement.name} = ${exprToString(statement.init)}`]; + case "assign": + return [`${pad}${exprToString(statement.target)} ${statement.op} ${exprToString(statement.value)}`]; + case "if": + return [ + `${pad}if ${exprToString(statement.test)}`, + ...statement.then.flatMap((item) => stmtToString(item, depth + 1)), + ...(statement.otherwise === undefined + ? [] + : [`${pad}else`, ...statement.otherwise.flatMap((item) => stmtToString(item, depth + 1))]), + ]; + case "switch": + return [ + `${pad}switch ${exprToString(statement.subject)}`, + ...statement.cases.flatMap((item) => [ + `${pad} case ${item.test === undefined ? "default" : exprToString(item.test)}`, + ...item.body.flatMap((inner) => stmtToString(inner, depth + 2)), + ]), + ]; + case "forCounted": + return [ + `${pad}for ${statement.name} from ${exprToString(statement.from)} to ${exprToString(statement.to)}`, + ...statement.body.flatMap((item) => stmtToString(item, depth + 1)), + ]; + case "forOf": + return [ + `${pad}for ${statement.name} of ${exprToString(statement.iterable)}`, + ...statement.body.flatMap((item) => stmtToString(item, depth + 1)), + ]; + case "return": + return [`${pad}return${statement.value === undefined ? "" : ` ${exprToString(statement.value)}`}`]; + case "throw": + return [`${pad}throw ${statement.errorClass}(${statement.args.map(exprToString).join(", ")})`]; + case "break": + return [`${pad}break`]; + case "continue": + return [`${pad}continue`]; + case "expr": + return [`${pad}${exprToString(statement.expr)}`]; + case "block": + return [`${pad}{`, ...statement.body.flatMap((item) => stmtToString(item, depth + 1)), `${pad}}`]; + default: { + const exhaustive: never = statement; + return exhaustive; + } + } +} diff --git a/engine/src/interp/interp.ts b/engine/src/interp/interp.ts new file mode 100644 index 000000000..72a9bdae8 --- /dev/null +++ b/engine/src/interp/interp.ts @@ -0,0 +1,379 @@ +/** + * The reference interpreter: the executable definition of the semantics. + * + * It runs the Core directly, so it is what every generated target is compared against, what + * comptime evaluation reuses, and what translation validation runs before and after each Core + * pass. Capabilities are injected, and the fakes are deterministic: a seeded PCG32, a virtual + * clock and scripted Http. + */ + +import { DomainFailure, lookupIntrinsic } from "../intrinsics/index.ts"; +import type { EvalContext } from "../intrinsics/index.ts"; +import type { CExpr, CFunc, CProgram, CStmt } from "../core/ir.ts"; +import { regexMatches } from "../regex.ts"; +import type { Value } from "../values.ts"; +import { + NONE, + asBigInt, + asList, + asRecord, + codePointsOf, + isNone, + record, + some, +} from "../values.ts"; + +export type HttpHandler = (request: { + method: string; + url: string; + headers: readonly { name: string; value: string }[]; + body: string; + timeoutMillis: bigint; +}) => + | { status: number; body: string; headers?: readonly { name: string; value: string }[]; latencyMillis?: number } + /** A transport error or a timeout, which the core reads as an absent response. */ + | undefined; + +export type Capabilities = { + /** Scripted Http. Throwing `DomainFailure("HttpError", …)` models a transport failure. */ + readonly http?: HttpHandler; + /** Virtual time in milliseconds; `clock.sleep` advances it instantly. */ + startMillis?: bigint; + /** Seed of the reference PCG32, so a Random-using utility is compared bit for bit. */ + readonly seed?: bigint; +}; + +type Signal = + | { readonly kind: "normal" } + | { readonly kind: "return"; readonly value: Value } + | { readonly kind: "break" } + | { readonly kind: "continue" }; + +const NORMAL: Signal = { kind: "normal" }; + +class VirtualClock { + now: bigint; + + constructor(start: bigint) { + this.now = start; + } +} + +/** The reference pseudo-random generator: PCG32 with the reference stream. */ +class Pcg32 { + private state: bigint; + private readonly increment: bigint; + + constructor(seed: bigint) { + this.state = 0n; + this.increment = 1442695040888963407n; + this.next(); + this.state = (this.state + seed) & 0xffff_ffff_ffff_ffffn; + this.next(); + } + + next(): bigint { + const previous = this.state; + this.state = (previous * 6364136223846793005n + this.increment) & 0xffff_ffff_ffff_ffffn; + const xorshifted = ((previous >> 18n) ^ previous) >> 27n & 0xffff_ffffn; + const rotation = previous >> 59n; + return ((xorshifted >> rotation) | (xorshifted << ((-rotation) & 31n))) & 0xffff_ffffn; + } +} + +export class Interpreter { + private readonly program: CProgram; + private readonly clock: VirtualClock; + private readonly random: Pcg32; + private readonly httpHandler: HttpHandler | undefined; + /** Set while a `race` task runs, so each task keeps its own virtual completion time. */ + private raceTime: { millis: bigint } | undefined; + + constructor(program: CProgram, capabilities: Capabilities = {}) { + this.program = program; + this.clock = new VirtualClock(capabilities.startMillis ?? 0n); + this.random = new Pcg32(capabilities.seed ?? 0x853c49e6748fea9bn); + this.httpHandler = capabilities.http; + } + + get virtualNow(): bigint { + return this.clock.now; + } + + call(name: string, args: readonly Value[]): Value { + const fn = this.program.functions.get(name); + if (fn === undefined) throw new Error(`unknown function ${name}`); + return this.invoke(fn, args); + } + + private invoke(fn: CFunc, args: readonly Value[]): Value { + const locals = new Map(); + fn.params.forEach((param, index) => locals.set(param.name, args[index]!)); + const signal = this.block(fn.body, locals); + return signal.kind === "return" ? signal.value : true; + } + + private get context(): EvalContext { + return { + http: (request) => { + if (this.httpHandler === undefined) { + throw new DomainFailure("HttpError", "no Http capability was provided"); + } + const fields = request.fields; + const response = this.httpHandler({ + method: String(fields["method"]), + url: String(fields["url"]), + headers: asList(fields["headers"]!).map((header) => { + const item = asRecord(header); + return { name: String(item.fields["name"]), value: String(item.fields["value"]) }; + }), + body: String(fields["body"]), + timeoutMillis: asBigInt(fields["timeoutMillis"]!), + }); + if (response === undefined) return undefined; + this.advance(BigInt(response.latencyMillis ?? 0)); + return record("HttpResponse", { + status: BigInt(response.status), + body: response.body, + headers: (response.headers ?? []).map((header) => + record("HttpHeader", { name: header.name, value: header.value }), + ), + }); + }, + now: () => this.currentMillis, + sleep: (milliseconds) => this.advance(milliseconds), + nextU32: () => this.random.next(), + }; + } + + private get currentMillis(): bigint { + return this.raceTime === undefined ? this.clock.now : this.raceTime.millis; + } + + private advance(milliseconds: bigint): void { + if (this.raceTime === undefined) this.clock.now += milliseconds; + else this.raceTime.millis += milliseconds; + } + + /* ---------------------------------------------------------------- * + * Statements + * ---------------------------------------------------------------- */ + + private block(body: readonly CStmt[], locals: Map): Signal { + for (const statement of body) { + const signal = this.statement(statement, locals); + if (signal.kind !== "normal") return signal; + } + return NORMAL; + } + + private statement(statement: CStmt, locals: Map): Signal { + switch (statement.kind) { + case "let": + locals.set(statement.name, this.expr(statement.init, locals)); + return NORMAL; + case "assign": + locals.set(statement.name, this.expr(statement.value, locals)); + return NORMAL; + case "setIndex": { + const list = [...asList(locals.get(statement.name)!)]; + list[Number(asBigInt(this.expr(statement.index, locals)))] = this.expr(statement.value, locals); + locals.set(statement.name, list); + return NORMAL; + } + case "push": { + const list = [...asList(locals.get(statement.name)!), this.expr(statement.value, locals)]; + locals.set(statement.name, list); + return NORMAL; + } + case "if": + return this.expr(statement.test, locals) === true + ? this.block(statement.then, locals) + : this.block(statement.otherwise, locals); + case "switch": { + const subject = this.expr(statement.subject, locals); + for (const entry of statement.cases) { + if (entry.values.some((value) => value === subject)) return this.block(entry.body, locals); + } + return statement.otherwise === undefined ? NORMAL : this.block(statement.otherwise, locals); + } + case "forRange": { + const from = asBigInt(this.expr(statement.from, locals)); + const to = asBigInt(this.expr(statement.to, locals)); + const step = statement.step; + for ( + let counter = from; + statement.step > 0n + ? statement.inclusive + ? counter <= to + : counter < to + : statement.inclusive + ? counter >= to + : counter > to; + counter += step + ) { + locals.set(statement.name, counter); + const signal = this.block(statement.body, locals); + if (signal.kind === "break") break; + if (signal.kind === "return") return signal; + } + return NORMAL; + } + case "forEach": { + for (const item of asList(this.expr(statement.iterable, locals))) { + locals.set(statement.name, item); + const signal = this.block(statement.body, locals); + if (signal.kind === "break") break; + if (signal.kind === "return") return signal; + } + return NORMAL; + } + case "return": + return { + kind: "return", + value: statement.value === undefined ? true : this.expr(statement.value, locals), + }; + case "fail": { + const message = statement.args.map((arg) => String(this.expr(arg, locals))).join(" "); + throw new DomainFailure(statement.errorClass, message); + } + case "break": + return { kind: "break" }; + case "continue": + return { kind: "continue" }; + case "expr": + this.expr(statement.expr, locals); + return NORMAL; + default: { + const exhaustive: never = statement; + return exhaustive; + } + } + } + + /* ---------------------------------------------------------------- * + * Expressions + * ---------------------------------------------------------------- */ + + expr(expr: CExpr, locals: Map): Value { + switch (expr.kind) { + case "lit": + return expr.value; + case "local": { + const value = locals.get(expr.name); + if (value === undefined) throw new Error(`unbound local ${expr.name}`); + return value; + } + case "none": + return NONE; + case "some": + return some(this.expr(expr.inner, locals)); + case "record": { + const fields: Record = {}; + for (const field of expr.fields) fields[field.name] = this.expr(field.value, locals); + return record(expr.typeName, fields); + } + case "field": { + const target = this.expr(expr.target, locals); + return asRecord(target).fields[expr.name]!; + } + case "list": + return expr.items.map((item) => this.expr(item, locals)); + case "call": { + const fn = this.program.functions.get(expr.fn); + if (fn === undefined) throw new Error(`unknown function ${expr.fn}`); + return this.invoke( + fn, + expr.args.map((arg) => this.expr(arg, locals)), + ); + } + case "op": + return this.operation(expr, locals); + case "lambda": + return { + __kind: "lambda", + call: (args: Value[]) => { + const inner = new Map(locals); + expr.params.forEach((param, index) => inner.set(param.name, args[index]!)); + const signal = this.block(expr.body, inner); + return signal.kind === "return" ? signal.value : true; + }, + }; + case "cond": + return this.expr(expr.test, locals) === true + ? this.expr(expr.then, locals) + : this.expr(expr.otherwise, locals); + case "and": + return this.expr(expr.left, locals) === true ? this.expr(expr.right, locals) : false; + case "or": + return this.expr(expr.left, locals) === true ? true : this.expr(expr.right, locals); + case "not": + return this.expr(expr.operand, locals) !== true; + default: { + const exhaustive: never = expr; + return exhaustive; + } + } + } + + private operation(expr: Extract, locals: Map): Value { + if (expr.op === "re.test") { + const subject = this.expr(expr.args[0]!, locals); + return regexMatches(expr.regex!, codePointsOf(String(subject))); + } + if (expr.op === "re.retain") { + const subject = String(this.expr(expr.args[0]!, locals)); + return codePointsOf(subject) + .filter((point) => regexMatches(expr.regex!, [point])) + .map((point) => String.fromCodePoint(point)) + .join(""); + } + if (expr.op === "task.race") { + return this.race(expr, locals); + } + const definition = lookupIntrinsic(expr.op); + if (definition === undefined) throw new Error(`unknown intrinsic ${expr.op}`); + const args = expr.args.map((arg) => this.expr(arg, locals)); + return definition.evaluate(args, this.context); + } + + /** + * `race` under the reference model. + * + * Each task runs under its own virtual clock, starting at the race's start time; the winner is + * the successful task with the smallest virtual completion time, ties broken by task index. + * Losing tasks are discarded, which is what makes cancellation unobservable. A task that fails + * with a domain error simply does not win, so `race` answers `none` only when all tasks fail. + */ + private race(expr: Extract, locals: Map): Value { + const tasks = asList(this.expr(expr.args[0]!, locals)); + const start = this.currentMillis; + const outcomes: { index: number; finishedAt: bigint; value: Value }[] = []; + const outer = this.raceTime; + for (const [index, task] of tasks.entries()) { + const scope = { millis: start }; + this.raceTime = scope; + try { + const value = (task as { call: (args: Value[]) => Value }).call([]); + if (!isNone(value)) outcomes.push({ index, finishedAt: scope.millis, value }); + } catch (error) { + if (!(error instanceof DomainFailure)) throw error; + } finally { + this.raceTime = outer; + } + } + if (outcomes.length === 0) return NONE; + outcomes.sort((left, right) => + left.finishedAt === right.finishedAt + ? left.index - right.index + : left.finishedAt < right.finishedAt + ? -1 + : 1, + ); + const winner = outcomes[0]!; + this.advance(winner.finishedAt - start); + return winner.value; + } +} + +export { DomainFailure, isNone }; diff --git a/engine/src/intrinsics/builtins.ts b/engine/src/intrinsics/builtins.ts new file mode 100644 index 000000000..3544aea79 --- /dev/null +++ b/engine/src/intrinsics/builtins.ts @@ -0,0 +1,52 @@ +/** + * Record and error types the engine itself defines, available to every project without being + * declared in source. They are the contract of the capability intrinsics. + */ + +import type { SemType } from "../types.ts"; +import { tInt, tList, tRecord, tString } from "../types.ts"; + +export type RecordDef = { + readonly name: string; + readonly fields: readonly { readonly name: string; readonly type: SemType; readonly optional: boolean }[]; + readonly doc: string; +}; + +function field(name: string, type: SemType, optional = false) { + return { name, type, optional }; +} + +export const BUILTIN_RECORDS: readonly RecordDef[] = [ + { + name: "HttpHeader", + doc: "One request or response header. Headers are an ordered list, never a map, so every target preserves order and duplicates.", + fields: [field("name", tString("ascii")), field("value", tString())], + }, + { + name: "HttpRequest", + doc: "A request handed to the Http capability. The host adds no retries and no hidden headers.", + fields: [ + field("method", tString("ascii", 3, 7)), + field("url", tString()), + field("headers", tList(tRecord("HttpHeader"))), + field("body", tString()), + field("timeoutMillis", tInt(0n, 600_000n)), + ], + }, + { + name: "HttpResponse", + doc: "A response from the Http capability. A status of 400 or more is a value, not a failure.", + fields: [ + field("status", tInt(0n, 599n)), + field("headers", tList(tRecord("HttpHeader"))), + field("body", tString()), + ], + }, +]; + +/** Domain errors the engine raises itself. Projects may catch them only by declaring them. */ +export const BUILTIN_ERRORS: readonly string[] = ["HttpError"]; + +export function builtinRecord(name: string): RecordDef | undefined { + return BUILTIN_RECORDS.find((record) => record.name === name); +} diff --git a/engine/src/intrinsics/capabilities.ts b/engine/src/intrinsics/capabilities.ts new file mode 100644 index 000000000..d9a04dd7b --- /dev/null +++ b/engine/src/intrinsics/capabilities.ts @@ -0,0 +1,140 @@ +/** + * The three capabilities — `http`, `clock` and `random` — and the one concurrency primitive. + * + * An author calls them directly and never mentions an environment; the compiler infers the effect + * and threads a capability record into exactly the functions that need one. Each target generates + * its own default implementation from its standard library, so there is no runtime package. + */ + +import { effects } from "../effects.ts"; +import { tDuration, tInstant, tInt, tOption, tRecord, tVoid } from "../types.ts"; +import { NONE, asBigInt, asRecord, some } from "../values.ts"; +import { SignatureError, defineIntrinsic, expectArity, expectKind } from "./registry.ts"; + +defineIntrinsic({ + name: "http.request", + paramHint: (index) => (index === 0 ? tRecord("HttpRequest") : undefined), + doc: "Performs one request. A transport error or a timeout answers `none`; a 4xx or 5xx status is a value, not a failure.", + effects: effects({ http: true }), + signature: (args) => { + expectArity("http.request", args, 1); + const request = expectKind("http.request", args, 0, "Record"); + if (request.name !== "HttpRequest") { + throw new SignatureError(`http.request: expects an HttpRequest, got ${request.name}`); + } + // Absence rather than an exception: the subset has no `catch`, so a retry or a fallback is + // written as ordinary control flow over an Option (docs/decisions/0006-http-is-an-option.md). + return tOption(tRecord("HttpResponse")); + }, + evaluate: ([request], ctx) => { + const response = ctx.http(asRecord(request!)); + return response === undefined ? NONE : some(response); + }, +}); + +defineIntrinsic({ + name: "clock.now", + doc: "The current instant, in milliseconds since the Unix epoch.", + effects: effects({ clock: true }), + signature: (args) => { + expectArity("clock.now", args, 0); + return tInstant; + }, + evaluate: (_args, ctx) => ctx.now(), +}); + +defineIntrinsic({ + name: "clock.sleep", + doc: "Suspends for a duration. Under the reference model this advances virtual time instantly.", + effects: effects({ clock: true }), + signature: (args) => { + expectArity("clock.sleep", args, 1); + expectKind("clock.sleep", args, 0, "Duration"); + return tVoid; + }, + evaluate: ([duration], ctx) => { + ctx.sleep(asBigInt(duration!)); + return true; + }, +}); + +defineIntrinsic({ + name: "clock.millis", + doc: "A duration from a count of milliseconds.", + signature: (args) => { + expectArity("clock.millis", args, 1); + const value = expectKind("clock.millis", args, 0, "Int"); + if (value.lo < 0n) throw new SignatureError("clock.millis: a duration may not be negative"); + return tDuration; + }, + evaluate: ([value]) => asBigInt(value!), +}); + +defineIntrinsic({ + name: "clock.elapsed", + doc: "The duration between two instants, `to - from`, clamped at zero.", + signature: (args) => { + expectArity("clock.elapsed", args, 2); + expectKind("clock.elapsed", args, 0, "Instant"); + expectKind("clock.elapsed", args, 1, "Instant"); + return tDuration; + }, + evaluate: ([from, to]) => { + const span = asBigInt(to!) - asBigInt(from!); + return span < 0n ? 0n : span; + }, +}); + +defineIntrinsic({ + name: "clock.durationMillis", + doc: "The millisecond count of a duration.", + signature: (args) => { + expectArity("clock.durationMillis", args, 1); + expectKind("clock.durationMillis", args, 0, "Duration"); + return tInt(0n, 2n ** 53n - 1n); + }, + evaluate: ([duration]) => asBigInt(duration!), +}); + +defineIntrinsic({ + name: "random.nextU32", + doc: "A uniform 32-bit value. Everything derived from it (ranges, shuffles) is written in source, so the algorithm is identical in every target.", + effects: effects({ random: true }), + signature: (args) => { + expectArity("random.nextU32", args, 0); + return tInt(0n, 2n ** 32n - 1n); + }, + evaluate: (_args, ctx) => ctx.nextU32(), +}); + +defineIntrinsic({ + name: "task.race", + doc: "Runs idempotent tasks concurrently and takes the first one to answer `some`, or `none` when none does.", + // The interpreter evaluates this one itself: it has to run each task under its own virtual + // clock to decide the winner deterministically (see docs/semantics.md, "Concurrency"). + effects: effects({}), + lambdaParams: () => [], + signature: (args) => { + expectArity("task.race", args, 1); + const tasks = expectKind("task.race", args, 0, "List"); + if (tasks.elem.kind !== "Lambda" || tasks.elem.params.length !== 0) { + throw new SignatureError( + "task.race: expects a list of zero-argument tasks", + "pass `[() => first(env), () => second(env)]`", + ); + } + if (tasks.elem.ret.kind !== "Option") { + throw new SignatureError( + "task.race: every task must answer an Option", + "a task that has nothing to report answers `undefined`, which is how a loser is recognized", + ); + } + if (tasks.min < 1) { + throw new SignatureError("task.race: the task list must be proven non-empty"); + } + return tasks.elem.ret; + }, + evaluate: () => { + throw new Error("task.race is evaluated by the interpreter, not by the registry"); + }, +}); diff --git a/engine/src/intrinsics/dates.ts b/engine/src/intrinsics/dates.ts new file mode 100644 index 000000000..12aa75df7 --- /dev/null +++ b/engine/src/intrinsics/dates.ts @@ -0,0 +1,194 @@ +/** + * `date`: the proleptic Gregorian calendar, years 1 to 9999, with no zone and no clock. + * + * `CivilDate` and `Instant` are never interchangeable: converting between them needs a time zone, + * which is deferred until a utility needs one. A host `Date` is the DX's problem, not the core's. + */ + +import { tBool, tCivilDate, tInt, tOption } from "../types.ts"; +import { NONE, asBigInt, asDate, civilDate, some } from "../values.ts"; +import { defineIntrinsic, expectArity, expectKind } from "./registry.ts"; + +export const MIN_EPOCH_DAY = -719_162; // 0001-01-01 +export const MAX_EPOCH_DAY = 2_932_896; // 9999-12-31 + +/** Days from the Unix epoch, after Howard Hinnant's `days_from_civil`. */ +export function epochDaysFromYmd(year: number, month: number, day: number): number { + const shifted = year - (month <= 2 ? 1 : 0); + const era = Math.floor(shifted / 400); + const yearOfEra = shifted - era * 400; + const dayOfYear = Math.floor((153 * (month + (month > 2 ? -3 : 9)) + 2) / 5) + day - 1; + const dayOfEra = yearOfEra * 365 + Math.floor(yearOfEra / 4) - Math.floor(yearOfEra / 100) + dayOfYear; + return era * 146_097 + dayOfEra - 719_468; +} + +export function ymdFromEpochDays(days: number): { year: number; month: number; day: number } { + const shifted = days + 719_468; + const era = Math.floor(shifted / 146_097); + const dayOfEra = shifted - era * 146_097; + const yearOfEra = Math.floor( + (dayOfEra - Math.floor(dayOfEra / 1460) + Math.floor(dayOfEra / 36_524) - Math.floor(dayOfEra / 146_096)) / 365, + ); + const year = yearOfEra + era * 400; + const dayOfYear = dayOfEra - (365 * yearOfEra + Math.floor(yearOfEra / 4) - Math.floor(yearOfEra / 100)); + const monthPrime = Math.floor((5 * dayOfYear + 2) / 153); + const day = dayOfYear - Math.floor((153 * monthPrime + 2) / 5) + 1; + const month = monthPrime + (monthPrime < 10 ? 3 : -9); + return { year: year + (month <= 2 ? 1 : 0), month, day }; +} + +export function isValidYmd(year: number, month: number, day: number): boolean { + if (year < 1 || year > 9999 || month < 1 || month > 12 || day < 1 || day > 31) return false; + const roundTrip = ymdFromEpochDays(epochDaysFromYmd(year, month, day)); + return roundTrip.year === year && roundTrip.month === month && roundTrip.day === day; +} + +defineIntrinsic({ + name: "date.fromYmd", + doc: "A civil date, or `none` when the components do not name a real day. Never rolls over.", + signature: (args) => { + expectArity("date.fromYmd", args, 3); + expectKind("date.fromYmd", args, 0, "Int"); + expectKind("date.fromYmd", args, 1, "Int"); + expectKind("date.fromYmd", args, 2, "Int"); + return tOption(tCivilDate); + }, + evaluate: ([year, month, day]) => { + const y = Number(asBigInt(year!)); + const m = Number(asBigInt(month!)); + const d = Number(asBigInt(day!)); + return isValidYmd(y, m, d) ? some(civilDate(epochDaysFromYmd(y, m, d))) : NONE; + }, +}); + +defineIntrinsic({ + name: "date.fromEpochDays", + doc: "A civil date from days since 1970-01-01, or `none` outside years 1 to 9999.", + signature: (args) => { + expectArity("date.fromEpochDays", args, 1); + expectKind("date.fromEpochDays", args, 0, "Int"); + return tOption(tCivilDate); + }, + evaluate: ([days]) => { + const value = Number(asBigInt(days!)); + return value < MIN_EPOCH_DAY || value > MAX_EPOCH_DAY ? NONE : some(civilDate(value)); + }, +}); + +defineIntrinsic({ + name: "date.clampEpochDays", + doc: "A civil date from days since 1970-01-01, clamped into years 1 to 9999. Total, so it needs no Option.", + signature: (args) => { + expectArity("date.clampEpochDays", args, 1); + expectKind("date.clampEpochDays", args, 0, "Int"); + return tCivilDate; + }, + evaluate: ([days]) => { + const value = Number(asBigInt(days!)); + return civilDate(Math.min(Math.max(value, MIN_EPOCH_DAY), MAX_EPOCH_DAY)); + }, +}); + +defineIntrinsic({ + name: "date.toEpochDays", + doc: "Days since 1970-01-01.", + signature: (args) => { + expectArity("date.toEpochDays", args, 1); + expectKind("date.toEpochDays", args, 0, "CivilDate"); + return tInt(BigInt(MIN_EPOCH_DAY), BigInt(MAX_EPOCH_DAY)); + }, + evaluate: ([date]) => BigInt(asDate(date!).days), +}); + +for (const [op, index] of [ + ["year", 0], + ["month", 1], + ["day", 2], +] as const) { + defineIntrinsic({ + name: `date.${op}`, + doc: `The ${op} component of a civil date.`, + signature: (args) => { + expectArity(`date.${op}`, args, 1); + expectKind(`date.${op}`, args, 0, "CivilDate"); + return index === 0 ? tInt(1n, 9999n) : index === 1 ? tInt(1n, 12n) : tInt(1n, 31n); + }, + evaluate: ([date]) => { + const parts = ymdFromEpochDays(asDate(date!).days); + return BigInt(index === 0 ? parts.year : index === 1 ? parts.month : parts.day); + }, + }); +} + +defineIntrinsic({ + name: "date.addDays", + doc: "Shifts by a number of days, or `none` when the result leaves years 1 to 9999.", + signature: (args) => { + expectArity("date.addDays", args, 2); + expectKind("date.addDays", args, 0, "CivilDate"); + expectKind("date.addDays", args, 1, "Int"); + return tOption(tCivilDate); + }, + evaluate: ([date, days]) => { + const shifted = asDate(date!).days + Number(asBigInt(days!)); + return shifted < MIN_EPOCH_DAY || shifted > MAX_EPOCH_DAY ? NONE : some(civilDate(shifted)); + }, +}); + +defineIntrinsic({ + name: "date.diffDays", + doc: "Exact day difference: `a - b`.", + signature: (args) => { + expectArity("date.diffDays", args, 2); + expectKind("date.diffDays", args, 0, "CivilDate"); + expectKind("date.diffDays", args, 1, "CivilDate"); + const span = BigInt(MAX_EPOCH_DAY - MIN_EPOCH_DAY); + return tInt(-span, span); + }, + evaluate: ([left, right]) => BigInt(asDate(left!).days - asDate(right!).days), +}); + +defineIntrinsic({ + name: "date.dayOfWeek", + doc: "ISO day of week: Monday is 1 through Sunday is 7.", + signature: (args) => { + expectArity("date.dayOfWeek", args, 1); + expectKind("date.dayOfWeek", args, 0, "CivilDate"); + return tInt(1n, 7n); + }, + evaluate: ([date]) => { + const days = asDate(date!).days; + // 1970-01-01 was a Thursday (ISO 4). + return BigInt(((((days + 3) % 7) + 7) % 7) + 1); + }, +}); + +defineIntrinsic({ + name: "date.compare", + doc: "Chronological comparison: -1, 0 or 1.", + signature: (args) => { + expectArity("date.compare", args, 2); + expectKind("date.compare", args, 0, "CivilDate"); + expectKind("date.compare", args, 1, "CivilDate"); + return tInt(-1n, 1n); + }, + evaluate: ([left, right]) => { + const a = asDate(left!).days; + const b = asDate(right!).days; + return a < b ? -1n : a > b ? 1n : 0n; + }, +}); + +defineIntrinsic({ + name: "date.isLeapYear", + doc: "Whether a year has 366 days in the proleptic Gregorian calendar.", + signature: (args) => { + expectArity("date.isLeapYear", args, 1); + expectKind("date.isLeapYear", args, 0, "Int"); + return tBool; + }, + evaluate: ([year]) => { + const value = Number(asBigInt(year!)); + return (value % 4 === 0 && value % 100 !== 0) || value % 400 === 0; + }, +}); diff --git a/engine/src/intrinsics/decimals.ts b/engine/src/intrinsics/decimals.ts new file mode 100644 index 000000000..098fa5431 --- /dev/null +++ b/engine/src/intrinsics/decimals.ts @@ -0,0 +1,306 @@ +/** + * `dec`: exact decimal arithmetic. + * + * Every operation that can lose information (division, rescale, conversion from a float) names + * its scale and its rounding mode at the call site. There is no hidden global context, which is + * what Python's 28-digit default context and Java's throwing `BigDecimal.divide` would otherwise + * impose on the generated code. + */ + +import type { SemType } from "../types.ts"; +import { tBool, tDecimal, tInt, typeToString } from "../types.ts"; +import { asBigInt, asDecimal, asNumber, asString, decimal } from "../values.ts"; +import { SignatureError, defineIntrinsic, expectArity, expectKind } from "./registry.ts"; + +export const ROUNDING_MODES = [ + "half-even", + "half-up", + "half-down", + "down", + "up", + "ceil", + "floor", +] as const; + +export type RoundingMode = (typeof ROUNDING_MODES)[number]; + +/** Rounds `numerator / denominator` (denominator > 0) to an integer under `mode`. */ +export function roundQuotient(numerator: bigint, denominator: bigint, mode: RoundingMode): bigint { + const negative = numerator < 0n; + const absolute = negative ? -numerator : numerator; + const quotient = absolute / denominator; + const remainder = absolute % denominator; + if (remainder === 0n) return negative ? -quotient : quotient; + + const twice = remainder * 2n; + let rounded: bigint; + switch (mode) { + case "down": + rounded = quotient; + break; + case "up": + rounded = quotient + 1n; + break; + case "ceil": + rounded = negative ? quotient : quotient + 1n; + break; + case "floor": + rounded = negative ? quotient + 1n : quotient; + break; + case "half-up": + rounded = twice >= denominator ? quotient + 1n : quotient; + break; + case "half-down": + rounded = twice > denominator ? quotient + 1n : quotient; + break; + case "half-even": + rounded = + twice > denominator || (twice === denominator && quotient % 2n === 1n) + ? quotient + 1n + : quotient; + break; + default: { + const exhaustive: never = mode; + return exhaustive; + } + } + return negative ? -rounded : rounded; +} + +/** The exact value of a binary64 as a rational, so conversions never go through a decimal string. */ +export function floatToRational(value: number): { numerator: bigint; denominator: bigint } { + if (!Number.isFinite(value)) throw new Error("float is not finite"); + const view = new DataView(new ArrayBuffer(8)); + view.setFloat64(0, value); + const bits = view.getBigUint64(0); + const sign = bits >> 63n === 1n ? -1n : 1n; + const exponent = Number((bits >> 52n) & 0x7ffn); + const mantissa = bits & 0xf_ffff_ffff_ffffn; + const significand = exponent === 0 ? mantissa : mantissa | (1n << 52n); + const power = (exponent === 0 ? 1 : exponent) - 1075; + if (power >= 0) { + return { numerator: sign * significand * 2n ** BigInt(power), denominator: 1n }; + } + return { numerator: sign * significand, denominator: 2n ** BigInt(-power) }; +} + +function constantScale(name: string, args: readonly SemType[], index: number): number { + const arg = expectKind(name, args, index, "Int"); + if (arg.lo !== arg.hi) { + throw new SignatureError( + `${name}: the scale must be a compile-time constant, got ${typeToString(arg)}`, + "pass a literal, for example 2", + ); + } + return Number(arg.lo); +} + +function roundingMode(name: string, args: readonly SemType[], index: number): void { + const arg = args[index]; + if (arg === undefined || arg.kind !== "Enum") { + throw new SignatureError(`${name}: argument ${index} must be a rounding mode`); + } + for (const member of arg.members) { + if (!(ROUNDING_MODES as readonly string[]).includes(member)) { + throw new SignatureError( + `${name}: ${JSON.stringify(member)} is not a rounding mode`, + `use one of ${ROUNDING_MODES.join(", ")}`, + ); + } + } +} + +defineIntrinsic({ + name: "dec.fromScaled", + doc: "A decimal from its unscaled integer and a constant scale: fromScaled(1234, 2) is 12.34.", + signature: (args) => { + expectArity("dec.fromScaled", args, 2); + expectKind("dec.fromScaled", args, 0, "Int"); + return tDecimal(constantScale("dec.fromScaled", args, 1)); + }, + evaluate: ([unscaled, scale]) => decimal(asBigInt(unscaled!), Number(asBigInt(scale!))), +}); + +defineIntrinsic({ + name: "dec.fromInt", + doc: "An exact decimal from an integer, at a constant scale.", + signature: (args) => { + expectArity("dec.fromInt", args, 2); + expectKind("dec.fromInt", args, 0, "Int"); + return tDecimal(constantScale("dec.fromInt", args, 1)); + }, + evaluate: ([value, scale]) => { + const places = Number(asBigInt(scale!)); + return decimal(asBigInt(value!) * 10n ** BigInt(places), places); + }, +}); + +defineIntrinsic({ + name: "dec.fromFloat", + doc: "Rounds the exact binary64 value to a constant scale under an explicit rounding mode.", + signature: (args) => { + expectArity("dec.fromFloat", args, 3); + expectKind("dec.fromFloat", args, 0, "Float"); + const scale = constantScale("dec.fromFloat", args, 1); + roundingMode("dec.fromFloat", args, 2); + return tDecimal(scale); + }, + evaluate: ([value, scale, mode]) => { + const places = Number(asBigInt(scale!)); + const { numerator, denominator } = floatToRational(asNumber(value!)); + const scaled = roundQuotient( + numerator * 10n ** BigInt(places), + denominator, + asString(mode!) as RoundingMode, + ); + return decimal(scaled, places); + }, +}); + +function sameScale(name: string, args: readonly SemType[]) { + const left = expectKind(name, args, 0, "Decimal"); + const right = expectKind(name, args, 1, "Decimal"); + if (left.scale !== right.scale) { + throw new SignatureError( + `${name}: both operands must have the same scale (${left.scale} vs ${right.scale})`, + "rescale one side explicitly with dec.rescale", + ); + } + return left; +} + +for (const op of ["add", "sub"] as const) { + defineIntrinsic({ + name: `dec.${op}`, + doc: `Exact decimal ${op}; both operands share a scale.`, + signature: (args) => { + expectArity(`dec.${op}`, args, 2); + return tDecimal(sameScale(`dec.${op}`, args).scale); + }, + evaluate: ([left, right]) => { + const a = asDecimal(left!); + const b = asDecimal(right!); + return decimal(op === "add" ? a.unscaled + b.unscaled : a.unscaled - b.unscaled, a.scale); + }, + }); +} + +defineIntrinsic({ + name: "dec.mul", + doc: "Exact decimal multiplication; the result's scale is the sum of the operands' scales.", + signature: (args) => { + expectArity("dec.mul", args, 2); + const left = expectKind("dec.mul", args, 0, "Decimal"); + const right = expectKind("dec.mul", args, 1, "Decimal"); + return tDecimal(left.scale + right.scale); + }, + evaluate: ([left, right]) => { + const a = asDecimal(left!); + const b = asDecimal(right!); + return decimal(a.unscaled * b.unscaled, a.scale + b.scale); + }, +}); + +defineIntrinsic({ + name: "dec.divRound", + doc: "Division to an explicit scale under an explicit rounding mode.", + signature: (args) => { + expectArity("dec.divRound", args, 4); + expectKind("dec.divRound", args, 0, "Decimal"); + expectKind("dec.divRound", args, 1, "Decimal"); + const scale = constantScale("dec.divRound", args, 2); + roundingMode("dec.divRound", args, 3); + return tDecimal(scale); + }, + evaluate: ([left, right, scale, mode]) => { + const a = asDecimal(left!); + const b = asDecimal(right!); + if (b.unscaled === 0n) throw new Error("dec.divRound: division by zero"); + const places = Number(asBigInt(scale!)); + const numerator = a.unscaled * 10n ** BigInt(places + b.scale); + const denominator = b.unscaled * 10n ** BigInt(a.scale); + const negativeDenominator = denominator < 0n; + const rounded = roundQuotient( + negativeDenominator ? -numerator : numerator, + negativeDenominator ? -denominator : denominator, + asString(mode!) as RoundingMode, + ); + return decimal(rounded, places); + }, +}); + +defineIntrinsic({ + name: "dec.rescale", + doc: "Changes the scale under an explicit rounding mode.", + signature: (args) => { + expectArity("dec.rescale", args, 3); + expectKind("dec.rescale", args, 0, "Decimal"); + const scale = constantScale("dec.rescale", args, 1); + roundingMode("dec.rescale", args, 2); + return tDecimal(scale); + }, + evaluate: ([value, scale, mode]) => { + const item = asDecimal(value!); + const places = Number(asBigInt(scale!)); + if (places >= item.scale) { + return decimal(item.unscaled * 10n ** BigInt(places - item.scale), places); + } + return decimal( + roundQuotient(item.unscaled, 10n ** BigInt(item.scale - places), asString(mode!) as RoundingMode), + places, + ); + }, +}); + +defineIntrinsic({ + name: "dec.compare", + doc: "Exact comparison: -1, 0 or 1.", + signature: (args) => { + expectArity("dec.compare", args, 2); + sameScale("dec.compare", args); + return tInt(-1n, 1n); + }, + evaluate: ([left, right]) => { + const a = asDecimal(left!).unscaled; + const b = asDecimal(right!).unscaled; + return a < b ? -1n : a > b ? 1n : 0n; + }, +}); + +defineIntrinsic({ + name: "dec.isNegative", + doc: "Whether the value is strictly below zero.", + signature: (args) => { + expectArity("dec.isNegative", args, 1); + expectKind("dec.isNegative", args, 0, "Decimal"); + return tBool; + }, + evaluate: ([value]) => asDecimal(value!).unscaled < 0n, +}); + +defineIntrinsic({ + name: "dec.abs", + doc: "Absolute value, keeping the scale.", + signature: (args) => { + expectArity("dec.abs", args, 1); + const value = expectKind("dec.abs", args, 0, "Decimal"); + return tDecimal(value.scale); + }, + evaluate: ([value]) => { + const item = asDecimal(value!); + return decimal(item.unscaled < 0n ? -item.unscaled : item.unscaled, item.scale); + }, +}); + +defineIntrinsic({ + name: "dec.unscaled", + doc: "The unscaled integer: dec.unscaled(12.34 at scale 2) is 1234.", + signature: (args) => { + expectArity("dec.unscaled", args, 1); + expectKind("dec.unscaled", args, 0, "Decimal"); + // A Decimal's magnitude is not tracked in its type, so the unscaled integer takes the + // platform-safe domain; a backend that needs more picks a wide representation. + return tInt(-(2n ** 53n - 1n), 2n ** 53n - 1n); + }, + evaluate: ([value]) => asDecimal(value!).unscaled, +}); diff --git a/engine/src/intrinsics/index.ts b/engine/src/intrinsics/index.ts new file mode 100644 index 000000000..fed285c4d --- /dev/null +++ b/engine/src/intrinsics/index.ts @@ -0,0 +1,24 @@ +/** Loads every intrinsic module, so importing this file populates the registry. */ + +import "./strings.ts"; +import "./numbers.ts"; +import "./sequences.ts"; +import "./decimals.ts"; +import "./dates.ts"; +import "./capabilities.ts"; +import "./regexes.ts"; +import "./options.ts"; + +export { allIntrinsics, lookupIntrinsic, SignatureError, DomainFailure } from "./registry.ts"; +export type { EvalContext, IntrinsicDef } from "./registry.ts"; +export { BUILTIN_ERRORS, BUILTIN_RECORDS, builtinRecord } from "./builtins.ts"; +export type { RecordDef } from "./builtins.ts"; +export { ROUNDING_MODES, roundQuotient, floatToRational } from "./decimals.ts"; +export { + epochDaysFromYmd, + ymdFromEpochDays, + isValidYmd, + MIN_EPOCH_DAY, + MAX_EPOCH_DAY, +} from "./dates.ts"; +export { TRIM_CODE_POINTS } from "./strings.ts"; diff --git a/engine/src/intrinsics/numbers.ts b/engine/src/intrinsics/numbers.ts new file mode 100644 index 000000000..8a929a784 --- /dev/null +++ b/engine/src/intrinsics/numbers.ts @@ -0,0 +1,246 @@ +/** + * `int`, `float` and the structural equality every type shares. + * + * Integers are mathematical: there is no wraparound anywhere, and every result carries the range + * proven from its operands, which is what lets a backend pick a representation it can prove safe. + * `/` and `%` are truncated (the sign of the dividend), the behavior of JavaScript, Go, Java and + * C#; the Python backend emits the equivalent because Python's `%` is floored. + */ + +import type { SemType } from "../types.ts"; +import { + rangeAdd, + rangeDiv, + rangeMod, + rangeMul, + rangeSub, + tBool, + tFloat, + tInt, + typeToString, +} from "../types.ts"; +import { asBigInt, asNumber, valuesEqual } from "../values.ts"; +import { SignatureError, defineIntrinsic, expectArity, expectKind } from "./registry.ts"; + +function intPair(name: string, args: readonly SemType[]) { + expectArity(name, args, 2); + return [expectKind(name, args, 0, "Int"), expectKind(name, args, 1, "Int")] as const; +} + +function floatPair(name: string, args: readonly SemType[]) { + expectArity(name, args, 2); + expectKind(name, args, 0, "Float"); + expectKind(name, args, 1, "Float"); +} + +const arithmetic: Record bigint> = { + add: (a, b) => a + b, + sub: (a, b) => a - b, + mul: (a, b) => a * b, +}; + +for (const [op, apply] of Object.entries(arithmetic)) { + defineIntrinsic({ + name: `int.${op}`, + doc: `Exact integer ${op}.`, + signature: (args) => { + const [left, right] = intPair(`int.${op}`, args); + const range = + op === "add" + ? rangeAdd(left, right) + : op === "sub" + ? rangeSub(left, right) + : rangeMul(left, right); + return tInt(range.lo, range.hi); + }, + evaluate: ([left, right]) => apply(asBigInt(left!), asBigInt(right!)), + }); +} + +function requireNonZero(name: string, divisor: Extract): void { + if (divisor.lo <= 0n && divisor.hi >= 0n) { + throw new SignatureError( + `${name}: the divisor may be zero (${typeToString(divisor)})`, + "guard the divisor, or derive it from a range that excludes zero", + ); + } +} + +defineIntrinsic({ + name: "int.div", + doc: "Truncated integer division. The divisor must be proven non-zero.", + signature: (args) => { + const [left, right] = intPair("int.div", args); + requireNonZero("int.div", right); + const range = rangeDiv(left, right); + return tInt(range.lo, range.hi); + }, + evaluate: ([left, right]) => asBigInt(left!) / asBigInt(right!), +}); + +defineIntrinsic({ + name: "int.mod", + doc: "Remainder with the sign of the dividend. The divisor must be proven non-zero.", + signature: (args) => { + const [left, right] = intPair("int.mod", args); + requireNonZero("int.mod", right); + const range = rangeMod(left, right); + return tInt(range.lo, range.hi); + }, + evaluate: ([left, right]) => asBigInt(left!) % asBigInt(right!), +}); + +defineIntrinsic({ + name: "int.neg", + doc: "Integer negation. Exact, like every integer operation: there is no wraparound.", + signature: (args) => { + expectArity("int.neg", args, 1); + const value = expectKind("int.neg", args, 0, "Int"); + return tInt(-value.hi, -value.lo); + }, + evaluate: ([value]) => -asBigInt(value!), +}); + +defineIntrinsic({ + name: "int.abs", + doc: "Absolute value, whose range the checker proves from the operand's.", + signature: (args) => { + expectArity("int.abs", args, 1); + const value = expectKind("int.abs", args, 0, "Int"); + const candidates = [value.lo < 0n ? -value.lo : value.lo, value.hi < 0n ? -value.hi : value.hi]; + const hi = candidates[0]! > candidates[1]! ? candidates[0]! : candidates[1]!; + const lo = value.lo <= 0n && value.hi >= 0n ? 0n : candidates[0]! < candidates[1]! ? candidates[0]! : candidates[1]!; + return tInt(lo, hi); + }, + evaluate: ([value]) => { + const number = asBigInt(value!); + return number < 0n ? -number : number; + }, +}); + +for (const op of ["min", "max"] as const) { + defineIntrinsic({ + name: `int.${op}`, + doc: `The ${op} of two integers.`, + signature: (args) => { + const [left, right] = intPair(`int.${op}`, args); + return op === "min" + ? tInt(left.lo < right.lo ? left.lo : right.lo, left.hi < right.hi ? left.hi : right.hi) + : tInt(left.lo > right.lo ? left.lo : right.lo, left.hi > right.hi ? left.hi : right.hi); + }, + evaluate: ([left, right]) => { + const a = asBigInt(left!); + const b = asBigInt(right!); + return op === "min" ? (a < b ? a : b) : a > b ? a : b; + }, + }); +} + +const comparisons: Record boolean> = { + lt: (ordering) => ordering < 0, + le: (ordering) => ordering <= 0, + gt: (ordering) => ordering > 0, + ge: (ordering) => ordering >= 0, +}; + +for (const [op, holds] of Object.entries(comparisons)) { + defineIntrinsic({ + name: `int.${op}`, + doc: `Integer comparison (${op}).`, + signature: (args) => { + intPair(`int.${op}`, args); + return tBool; + }, + evaluate: ([left, right]) => { + const a = asBigInt(left!); + const b = asBigInt(right!); + return holds(a < b ? -1 : a > b ? 1 : 0); + }, + }); + + defineIntrinsic({ + name: `float.${op}`, + doc: `Float comparison (${op}).`, + signature: (args) => { + floatPair(`float.${op}`, args); + return tBool; + }, + evaluate: ([left, right]) => { + const a = asNumber(left!); + const b = asNumber(right!); + return holds(a < b ? -1 : a > b ? 1 : 0); + }, + }); +} + +const floatArithmetic: Record number> = { + add: (a, b) => a + b, + sub: (a, b) => a - b, + mul: (a, b) => a * b, + div: (a, b) => a / b, +}; + +for (const [op, apply] of Object.entries(floatArithmetic)) { + defineIntrinsic({ + name: `float.${op}`, + doc: `IEEE-754 binary64 ${op}, correctly rounded in every target.`, + signature: (args) => { + floatPair(`float.${op}`, args); + return tFloat; + }, + evaluate: ([left, right]) => apply(asNumber(left!), asNumber(right!)), + }); +} + +defineIntrinsic({ + name: "float.neg", + doc: "Float negation, which preserves the sign of zero.", + signature: (args) => { + expectArity("float.neg", args, 1); + expectKind("float.neg", args, 0, "Float"); + return tFloat; + }, + evaluate: ([value]) => -asNumber(value!), +}); + +defineIntrinsic({ + name: "float.fromInt", + doc: "Exact conversion of an integer whose range fits binary64 without rounding.", + signature: (args) => { + expectArity("float.fromInt", args, 1); + const value = expectKind("float.fromInt", args, 0, "Int"); + const limit = 2n ** 53n; + if (value.lo < -limit || value.hi > limit) { + throw new SignatureError( + `float.fromInt: ${typeToString(value)} does not convert exactly to binary64`, + "narrow the integer's range, or keep the value as a Decimal", + ); + } + return tFloat; + }, + evaluate: ([value]) => Number(asBigInt(value!)), +}); + +defineIntrinsic({ + name: "core.eq", + doc: "Structural equality. Both sides must have the same shape.", + signature: (args) => { + expectArity("core.eq", args, 2); + const left = args[0]!; + const right = args[1]!; + if (left.kind !== right.kind && left.kind !== "Never" && right.kind !== "Never") { + throw new SignatureError( + `cannot compare ${typeToString(left)} with ${typeToString(right)}`, + "compare values of the same type; there is no implicit conversion", + ); + } + if (left.kind === "Float") { + throw new SignatureError( + "floats are not compared with ===", + "compare with an explicit tolerance, or use Decimal", + ); + } + return tBool; + }, + evaluate: ([left, right]) => valuesEqual(left!, right!), +}); diff --git a/engine/src/intrinsics/options.ts b/engine/src/intrinsics/options.ts new file mode 100644 index 000000000..f64000dcd --- /dev/null +++ b/engine/src/intrinsics/options.ts @@ -0,0 +1,58 @@ +/** + * `opt`: explicit absence. + * + * `opt.unwrap` is never written by an author: the checker emits it where flow analysis has proven + * the value present, which is why there is no unchecked access anywhere in generated code. + */ + +import { tBool, tOption, typeToString } from "../types.ts"; +import { NONE, isNone, unwrap } from "../values.ts"; +import { SignatureError, defineIntrinsic, expectArity, expectKind } from "./registry.ts"; + +defineIntrinsic({ + name: "opt.isNone", + doc: "Whether the option is absent. This is what `x === undefined` means in the Core.", + signature: (args) => { + expectArity("opt.isNone", args, 1); + expectKind("opt.isNone", args, 0, "Option"); + return tBool; + }, + evaluate: ([value]) => isNone(value!), +}); + +defineIntrinsic({ + name: "opt.unwrap", + doc: "The value inside an option the checker has proven present.", + signature: (args) => { + expectArity("opt.unwrap", args, 1); + const option = args[0]!; + if (option.kind !== "Option") { + throw new SignatureError(`opt.unwrap: expects an Option, got ${typeToString(option)}`); + } + return option.inner; + }, + evaluate: ([value]) => unwrap(value!), +}); + +defineIntrinsic({ + name: "opt.some", + doc: "Wraps a present value.", + signature: (args) => { + expectArity("opt.some", args, 1); + return tOption(args[0]!); + }, + evaluate: ([value]) => ({ __kind: "some", value: value! }), +}); + +defineIntrinsic({ + name: "opt.orElse", + doc: "The value, or a default when absent. This is what `??` means in the Core.", + signature: (args) => { + expectArity("opt.orElse", args, 2); + const option = expectKind("opt.orElse", args, 0, "Option"); + return option.inner; + }, + evaluate: ([value, fallback]) => (isNone(value!) ? fallback! : unwrap(value!)), +}); + +export const OPTION_NONE = NONE; diff --git a/engine/src/intrinsics/regexes.ts b/engine/src/intrinsics/regexes.ts new file mode 100644 index 000000000..a132c4090 --- /dev/null +++ b/engine/src/intrinsics/regexes.ts @@ -0,0 +1,38 @@ +/** + * `re`: regex matching over a comptime pattern. + * + * The pattern is never a run-time value. The frontend requires a literal (or a constant bound to + * one), normalizes it at compile time, and attaches the normalized form to the Core node as a + * payload, which is why this intrinsic takes only the subject. + */ + +import { MAX_COLLECTION_LENGTH, tBool, tString } from "../types.ts"; +import { defineIntrinsic, expectArity, expectKind } from "./registry.ts"; + +defineIntrinsic({ + name: "re.retain", + doc: "Keeps only the scalars matching a comptime character class, dropping everything else. The class also refines the result.", + signature: (args) => { + expectArity("re.retain", args, 1); + const text = expectKind("re.retain", args, 0, "String"); + // The result's class comes from the pattern, which the frontend attaches as a payload; the + // checker narrows it there, because only it knows which class the literal denoted. + return tString("none", 0, Math.min(text.max, MAX_COLLECTION_LENGTH)); + }, + evaluate: () => { + throw new Error("re.retain is evaluated with its pattern payload by the interpreter"); + }, +}); + +defineIntrinsic({ + name: "re.test", + doc: "Whether the whole string matches the pattern. Always a full match; there are no partial matches in the subset.", + signature: (args) => { + expectArity("re.test", args, 1); + expectKind("re.test", args, 0, "String"); + return tBool; + }, + evaluate: () => { + throw new Error("re.test is evaluated with its pattern payload by the interpreter"); + }, +}); diff --git a/engine/src/intrinsics/registry.ts b/engine/src/intrinsics/registry.ts new file mode 100644 index 000000000..dc05061f0 --- /dev/null +++ b/engine/src/intrinsics/registry.ts @@ -0,0 +1,138 @@ +/** + * The intrinsic registry: one specification per operation, shared by the checker, the reference + * interpreter, the comptime evaluator and every backend. + * + * An intrinsic exists only when it passes the admission rule (docs/semantics.md, "Admission"): + * it cannot be expressed efficiently and idiomatically as source library code, it has a precise + * spec with a reference implementation and vectors, and at least two utilities need it or it is a + * prerequisite of one that is admitted. + */ + +import type { EffectSet } from "../effects.ts"; +import { PURE } from "../effects.ts"; +import type { SemType } from "../types.ts"; +import { typeToString } from "../types.ts"; +import type { RecordValue, Value } from "../values.ts"; + +/** Raised by a signature when the call does not type check, or lacks a required fact. */ +export class SignatureError extends Error { + readonly suggestion: string | undefined; + + constructor(message: string, suggestion?: string) { + super(message); + this.name = "SignatureError"; + this.suggestion = suggestion; + } +} + +/** A domain failure raised by `throw` in source, or by a failing intrinsic. */ +export class DomainFailure extends Error { + readonly errorType: string; + + constructor(errorType: string, message: string) { + super(message); + this.name = "DomainFailure"; + this.errorType = errorType; + } +} + +/** The capabilities the reference interpreter hands to effectful intrinsics. */ +export type EvalContext = { + /** `undefined` models a transport error or a timeout. */ + readonly http: (request: RecordValue) => RecordValue | undefined; + readonly now: () => bigint; + readonly sleep: (milliseconds: bigint) => void; + readonly nextU32: () => bigint; +}; + +export type IntrinsicDef = { + readonly name: string; + readonly doc: string; + readonly effects: EffectSet; + /** Computes the result type, or throws `SignatureError` with a span-free explanation. */ + readonly signature: (args: readonly SemType[]) => SemType; + readonly evaluate: (args: readonly Value[], ctx: EvalContext) => Value; + /** False when the result may not be baked at compile time (capabilities). */ + readonly comptime: boolean; + /** + * Parameter types of a lambda argument, given the types of the arguments before it. Combinators + * declare this so the checker can type a lambda body against the element type it will receive, + * which is why authors never annotate a combinator's lambda. + */ + readonly lambdaParams?: (prior: readonly SemType[], index: number) => readonly SemType[]; + /** + * The type an argument is expected to have, which lets a record literal be written inline at + * the call site the way it is for a declared function. + */ + readonly paramHint?: (index: number, prior: readonly SemType[]) => SemType | undefined; +}; + +const registry = new Map(); + +export function defineIntrinsic( + def: Omit & { + effects?: EffectSet; + comptime?: boolean; + }, +): IntrinsicDef { + const effects = def.effects ?? PURE; + const full: IntrinsicDef = { + ...def, + effects, + comptime: def.comptime ?? (effects.http || effects.clock || effects.random ? false : true), + }; + if (registry.has(full.name)) throw new Error(`duplicate intrinsic ${full.name}`); + registry.set(full.name, full); + return full; +} + +export function lookupIntrinsic(name: string): IntrinsicDef | undefined { + return registry.get(name); +} + +export function allIntrinsics(): readonly IntrinsicDef[] { + return [...registry.values()].sort((a, b) => a.name.localeCompare(b.name)); +} + +/* ------------------------------------------------------------------ * + * Signature helpers + * ------------------------------------------------------------------ */ + +export function expectArity(name: string, args: readonly SemType[], arity: number): void { + if (args.length !== arity) { + throw new SignatureError(`${name} takes ${arity} argument(s), got ${args.length}`); + } +} + +export function expectKind( + name: string, + args: readonly SemType[], + index: number, + kind: K, +): Extract { + const arg = args[index]; + if (arg === undefined || arg.kind !== kind) { + throw new SignatureError( + `${name}: argument ${index} must be ${kind}, got ${arg === undefined ? "nothing" : typeToString(arg)}`, + ); + } + return arg as Extract; +} + +/** Requires a proven character class on a string argument; this is how facts gate lowerings. */ +export function expectStringClass( + name: string, + args: readonly SemType[], + index: number, + cls: "ascii" | "digits", +): Extract { + const arg = expectKind(name, args, index, "String"); + const ok = cls === "ascii" ? arg.cls !== "none" : arg.cls === "digits"; + if (!ok) { + throw new SignatureError( + `${name}: argument ${index} must be proven ${cls}, got ${typeToString(arg)}`, + `narrow it first, for example with a regex guard (\`if (!re.test(PATTERN, value)) …\`) or \`str.as${cls === "ascii" ? "Ascii" : "Digits"}\``, + ); + } + return arg; +} diff --git a/engine/src/intrinsics/sequences.ts b/engine/src/intrinsics/sequences.ts new file mode 100644 index 000000000..e8f5e49aa --- /dev/null +++ b/engine/src/intrinsics/sequences.ts @@ -0,0 +1,341 @@ +/** + * `seq`: combinators over immutable lists. + * + * The Core never chooses a target idiom, so an author may write either a combinator or an + * imperative loop; the loop raiser turns recognizable loops into these, and each backend decides + * whether to print a comprehension, a native array method or a plain loop. + */ + +import type { SemType } from "../types.ts"; +import { + MAX_COLLECTION_LENGTH, + isSubtype, + join, + rangeMul, + tBool, + tFloat, + tInt, + tList, + tOption, + typeToString, +} from "../types.ts"; +import type { Value } from "../values.ts"; +import { NONE, asBigInt, asLambda, asList, asNumber, some, valuesEqual } from "../values.ts"; +import { SignatureError, defineIntrinsic, expectArity, expectKind } from "./registry.ts"; + +function expectList(name: string, args: readonly SemType[], index: number) { + return expectKind(name, args, index, "List"); +} + +function expectLambda(name: string, args: readonly SemType[], index: number) { + const arg = args[index]; + if (arg === undefined || arg.kind !== "Lambda") { + throw new SignatureError(`${name}: argument ${index} must be a function`); + } + return arg; +} + +defineIntrinsic({ + name: "seq.len", + doc: "Number of elements.", + signature: (args) => { + expectArity("seq.len", args, 1); + const list = expectList("seq.len", args, 0); + return tInt(BigInt(list.min), BigInt(list.max)); + }, + evaluate: ([list]) => BigInt(asList(list!).length), +}); + +defineIntrinsic({ + name: "seq.get", + doc: "Element at an index proven to be in range.", + signature: (args) => { + expectArity("seq.get", args, 2); + const list = expectList("seq.get", args, 0); + const index = expectKind("seq.get", args, 1, "Int"); + if (index.lo < 0n || index.hi >= BigInt(list.min)) { + throw new SignatureError( + `seq.get: the index may be out of range (index ${typeToString(index)} into ${typeToString(list)})`, + "guard the index against the list length, or narrow the list's length range", + ); + } + return list.elem; + }, + evaluate: ([list, index]) => asList(list!)[Number(asBigInt(index!))]!, +}); + +defineIntrinsic({ + name: "seq.at", + doc: "The element at an index, or `none` when the index is outside the list. The checked form of seq.get.", + signature: (args) => { + expectArity("seq.at", args, 2); + const list = expectList("seq.at", args, 0); + expectKind("seq.at", args, 1, "Int"); + return tOption(list.elem); + }, + evaluate: ([list, index]) => { + const items = asList(list!); + const position = Number(asBigInt(index!)); + return position < 0 || position >= items.length ? NONE : some(items[position]!); + }, +}); + +defineIntrinsic({ + name: "seq.map", + doc: "Applies a pure function to every element.", + lambdaParams: (prior) => [expectList("seq.map", prior, 0).elem], + signature: (args) => { + expectArity("seq.map", args, 2); + const list = expectList("seq.map", args, 0); + const fn = expectLambda("seq.map", args, 1); + return tList(fn.ret, list.min, list.max); + }, + evaluate: ([list, fn]) => asList(list!).map((item) => asLambda(fn!).call([item])), +}); + +defineIntrinsic({ + name: "seq.filter", + doc: "Keeps the elements a pure predicate accepts.", + lambdaParams: (prior) => [expectList("seq.filter", prior, 0).elem], + signature: (args) => { + expectArity("seq.filter", args, 2); + const list = expectList("seq.filter", args, 0); + const fn = expectLambda("seq.filter", args, 1); + if (fn.ret.kind !== "Bool") throw new SignatureError("seq.filter: the predicate must return Bool"); + return tList(list.elem, 0, list.max); + }, + evaluate: ([list, fn]) => asList(list!).filter((item) => asLambda(fn!).call([item]) === true), +}); + +defineIntrinsic({ + name: "seq.fold", + doc: "Left fold with an explicit initial value.", + lambdaParams: (prior) => [prior[1]!, expectList("seq.fold", prior, 0).elem], + signature: (args) => { + expectArity("seq.fold", args, 3); + expectList("seq.fold", args, 0); + const initial = args[1]!; + const fn = expectLambda("seq.fold", args, 2); + return join(initial, fn.ret); + }, + evaluate: ([list, initial, fn]) => + asList(list!).reduce((accumulator, item) => asLambda(fn!).call([accumulator, item]), initial!), +}); + +defineIntrinsic({ + name: "seq.sum", + doc: "Sum of a list of integers or floats.", + signature: (args) => { + expectArity("seq.sum", args, 1); + const list = expectList("seq.sum", args, 0); + if (list.elem.kind === "Float") return tFloat; + if (list.elem.kind !== "Int") throw new SignatureError("seq.sum: expects Int or Float elements"); + const total = rangeMul( + { lo: list.elem.lo, hi: list.elem.hi }, + { lo: BigInt(list.min), hi: BigInt(list.max) }, + ); + return tInt(total.lo < 0n ? total.lo : 0n, total.hi > 0n ? total.hi : 0n); + }, + evaluate: ([list]) => { + const items = asList(list!); + if (items.length > 0 && typeof items[0] === "number") { + return items.reduce((total, item) => total + asNumber(item), 0); + } + return items.reduce((total, item) => total + asBigInt(item), 0n); + }, +}); + +for (const op of ["any", "all"] as const) { + defineIntrinsic({ + name: `seq.${op}`, + doc: `True when ${op === "any" ? "at least one element" : "every element"} satisfies the predicate.`, + lambdaParams: (prior) => [expectList(`seq.${op}`, prior, 0).elem], + signature: (args) => { + expectArity(`seq.${op}`, args, 2); + expectList(`seq.${op}`, args, 0); + const fn = expectLambda(`seq.${op}`, args, 1); + if (fn.ret.kind !== "Bool") throw new SignatureError(`seq.${op}: the predicate must return Bool`); + return tBool; + }, + evaluate: ([list, fn]) => + op === "any" + ? asList(list!).some((item) => asLambda(fn!).call([item]) === true) + : asList(list!).every((item) => asLambda(fn!).call([item]) === true), + }); +} + +defineIntrinsic({ + name: "seq.find", + doc: "The first element satisfying the predicate, or `none`.", + lambdaParams: (prior) => [expectList("seq.find", prior, 0).elem], + signature: (args) => { + expectArity("seq.find", args, 2); + const list = expectList("seq.find", args, 0); + const fn = expectLambda("seq.find", args, 1); + if (fn.ret.kind !== "Bool") throw new SignatureError("seq.find: the predicate must return Bool"); + return tOption(list.elem); + }, + evaluate: ([list, fn]) => { + const found = asList(list!).find((item) => asLambda(fn!).call([item]) === true); + return found === undefined ? NONE : some(found); + }, +}); + +defineIntrinsic({ + name: "seq.indexOf", + doc: "Index of the first structurally equal element, or -1.", + signature: (args) => { + expectArity("seq.indexOf", args, 2); + const list = expectList("seq.indexOf", args, 0); + if (!isSubtype(args[1]!, list.elem) && !isSubtype(list.elem, args[1]!)) { + throw new SignatureError( + `seq.indexOf: cannot look for ${typeToString(args[1]!)} in ${typeToString(list)}`, + ); + } + return tInt(-1n, BigInt(Math.max(0, list.max - 1))); + }, + evaluate: ([list, needle]) => + BigInt(asList(list!).findIndex((item) => valuesEqual(item, needle!))), +}); + +defineIntrinsic({ + name: "seq.contains", + doc: "Whether a structurally equal element is present.", + signature: (args) => { + expectArity("seq.contains", args, 2); + const list = expectList("seq.contains", args, 0); + if (!isSubtype(args[1]!, list.elem) && !isSubtype(list.elem, args[1]!)) { + throw new SignatureError( + `seq.contains: cannot look for ${typeToString(args[1]!)} in ${typeToString(list)}`, + ); + } + return tBool; + }, + evaluate: ([list, needle]) => asList(list!).some((item) => valuesEqual(item, needle!)), +}); + +defineIntrinsic({ + name: "seq.concat", + doc: "Concatenation of two lists.", + signature: (args) => { + expectArity("seq.concat", args, 2); + const left = expectList("seq.concat", args, 0); + const right = expectList("seq.concat", args, 1); + return tList( + join(left.elem, right.elem), + Math.min(left.min + right.min, MAX_COLLECTION_LENGTH), + Math.min(left.max + right.max, MAX_COLLECTION_LENGTH), + ); + }, + evaluate: ([left, right]) => [...asList(left!), ...asList(right!)], +}); + +defineIntrinsic({ + name: "seq.slice", + doc: "A slice, clamped to the list's length, from inclusive to exclusive.", + signature: (args) => { + expectArity("seq.slice", args, 3); + const list = expectList("seq.slice", args, 0); + const from = expectKind("seq.slice", args, 1, "Int"); + const to = expectKind("seq.slice", args, 2, "Int"); + if (from.lo < 0n || to.lo < 0n) throw new SignatureError("seq.slice: negative bounds are outside the subset"); + const max = Math.min(Number(to.hi - from.lo), list.max); + return tList(list.elem, Math.max(0, Math.min(Number(to.lo - from.hi), max)), Math.max(0, max)); + }, + evaluate: ([list, from, to]) => + asList(list!).slice(Number(asBigInt(from!)), Number(asBigInt(to!))), +}); + +defineIntrinsic({ + name: "seq.reverse", + doc: "The list in reverse order.", + signature: (args) => { + expectArity("seq.reverse", args, 1); + const list = expectList("seq.reverse", args, 0); + return tList(list.elem, list.min, list.max); + }, + evaluate: ([list]) => [...asList(list!)].reverse(), +}); + +defineIntrinsic({ + name: "seq.sortStable", + doc: "Stable sort by an explicit comparator returning a negative, zero or positive Int.", + lambdaParams: (prior) => { + const elem = expectList("seq.sortStable", prior, 0).elem; + return [elem, elem]; + }, + signature: (args) => { + expectArity("seq.sortStable", args, 2); + const list = expectList("seq.sortStable", args, 0); + const fn = expectLambda("seq.sortStable", args, 1); + if (fn.ret.kind !== "Int") { + throw new SignatureError("seq.sortStable: the comparator must return an Int"); + } + return tList(list.elem, list.min, list.max); + }, + evaluate: ([list, fn]) => { + // Explicitly stable: decorate with the original index and break ties by it, so the + // reference never depends on the host sort's stability. + const decorated = asList(list!).map((item, index) => ({ item, index })); + decorated.sort((left, right) => { + const ordering = Number(asBigInt(asLambda(fn!).call([left.item, right.item]))); + return ordering !== 0 ? ordering : left.index - right.index; + }); + return decorated.map((entry) => entry.item); + }, +}); + +defineIntrinsic({ + name: "seq.sortStableBy", + doc: "Stable sort by a key: Int, Decimal, CivilDate or a string compared in scalar order.", + lambdaParams: (prior) => [expectList("seq.sortStableBy", prior, 0).elem], + signature: (args) => { + expectArity("seq.sortStableBy", args, 2); + const list = expectList("seq.sortStableBy", args, 0); + const fn = expectLambda("seq.sortStableBy", args, 1); + const allowed = ["Int", "String", "Decimal", "CivilDate"]; + if (!allowed.includes(fn.ret.kind)) { + throw new SignatureError( + `seq.sortStableBy: the key must be one of ${allowed.join(", ")}, got ${typeToString(fn.ret)}`, + "return an Int or a String key, or use seq.sortStable with an explicit comparator", + ); + } + return tList(list.elem, list.min, list.max); + }, + evaluate: ([list, fn]) => { + const decorated = asList(list!).map((item, index) => ({ + item, + index, + key: asLambda(fn!).call([item]), + })); + decorated.sort((left, right) => { + const ordering = compareKeys(left.key, right.key); + return ordering !== 0 ? ordering : left.index - right.index; + }); + return decorated.map((entry) => entry.item); + }, +}); + +function compareKeys(left: Value, right: Value): number { + if (typeof left === "bigint" && typeof right === "bigint") { + return left < right ? -1 : left > right ? 1 : 0; + } + if (typeof left === "string" && typeof right === "string") { + const a = [...left].map((scalar) => scalar.codePointAt(0)!); + const b = [...right].map((scalar) => scalar.codePointAt(0)!); + const shared = Math.min(a.length, b.length); + for (let index = 0; index < shared; index++) { + if (a[index]! !== b[index]!) return a[index]! < b[index]! ? -1 : 1; + } + return a.length - b.length; + } + if (typeof left === "object" && typeof right === "object" && left !== null && right !== null) { + const a = left as { __kind: string; unscaled?: bigint; days?: number }; + const b = right as { __kind: string; unscaled?: bigint; days?: number }; + if (a.__kind === "decimal" && b.__kind === "decimal") { + return a.unscaled! < b.unscaled! ? -1 : a.unscaled! > b.unscaled! ? 1 : 0; + } + if (a.__kind === "date" && b.__kind === "date") return a.days! - b.days!; + } + throw new Error("unsupported sort key"); +} diff --git a/engine/src/intrinsics/strings.ts b/engine/src/intrinsics/strings.ts new file mode 100644 index 000000000..22ab8e8fb --- /dev/null +++ b/engine/src/intrinsics/strings.ts @@ -0,0 +1,435 @@ +/** + * `str`: operations on immutable sequences of Unicode scalars. + * + * Positional operations (`charAt`, `codeAt`, `slice`) are admitted only on `Ascii`, where the + * index means the same thing in every target: a byte in Go, a `str` position in Python, a code + * unit in TypeScript. Generic `String` supports iteration by scalar instead, so no target ever + * has to reproduce UTF-16 indexing. + */ + +import { + MAX_COLLECTION_LENGTH, + tInt, + tList, + tOption, + tString, + typeToString, + weakerClass, +} from "../types.ts"; +import type { SemType } from "../types.ts"; +import { + NONE, + asBigInt, + asList, + asString, + codePointsOf, + compareScalars, + fromCodePoints, + some, +} from "../values.ts"; +import { SignatureError, defineIntrinsic, expectArity, expectKind, expectStringClass } from "./registry.ts"; + +const SCALAR = () => tInt(0n, 0x10ffffn); + +/** Proves that `index` can only address a scalar that `text` certainly has. */ +function proveIndex(name: string, text: Extract, index: SemType): void { + if (index.kind !== "Int") throw new SignatureError(`${name}: the index must be an Int`); + if (index.lo < 0n) { + throw new SignatureError( + `${name}: the index may be negative (${typeToString(index)})`, + "guard the index, or derive it from a proven length", + ); + } + if (index.hi >= BigInt(text.min)) { + throw new SignatureError( + `${name}: the index may address past the end of a ${typeToString(text)} (index ${typeToString(index)})`, + "narrow the string's length first, for example with a regex guard or a `length` check", + ); + } +} + +defineIntrinsic({ + name: "str.len", + doc: "Number of Unicode scalars in the string.", + signature: (args) => { + expectArity("str.len", args, 1); + const text = expectKind("str.len", args, 0, "String"); + return tInt(BigInt(text.min), BigInt(text.max)); + }, + evaluate: ([text]) => BigInt(codePointsOf(asString(text!)).length), +}); + +defineIntrinsic({ + name: "str.codePoints", + doc: "The scalars of the string, as code points.", + signature: (args) => { + expectArity("str.codePoints", args, 1); + const text = expectKind("str.codePoints", args, 0, "String"); + const elem = + text.cls === "digits" ? tInt(0x30n, 0x39n) : text.cls === "ascii" ? tInt(0n, 0x7fn) : SCALAR(); + return tList(elem, text.min, text.max); + }, + evaluate: ([text]) => codePointsOf(asString(text!)).map((point) => BigInt(point)), +}); + +defineIntrinsic({ + name: "str.fromCodePoints", + doc: "Builds a string from code points.", + signature: (args) => { + expectArity("str.fromCodePoints", args, 1); + const list = expectKind("str.fromCodePoints", args, 0, "List"); + const elem = list.elem; + if (elem.kind !== "Int") throw new SignatureError("str.fromCodePoints: expects a list of Int"); + const cls = + elem.lo >= 0x30n && elem.hi <= 0x39n ? "digits" : elem.hi <= 0x7fn ? "ascii" : "none"; + return tString(cls, list.min, list.max); + }, + evaluate: ([list]) => fromCodePoints(asList(list!).map((point) => Number(asBigInt(point)))), +}); + +defineIntrinsic({ + name: "str.concat", + doc: "Concatenation. Also the lowering of `+` on strings.", + signature: (args) => { + expectArity("str.concat", args, 2); + const left = expectKind("str.concat", args, 0, "String"); + const right = expectKind("str.concat", args, 1, "String"); + return tString( + weakerClass(left.cls, right.cls), + Math.min(left.min + right.min, MAX_COLLECTION_LENGTH), + Math.min(left.max + right.max, MAX_COLLECTION_LENGTH), + ); + }, + evaluate: ([left, right]) => `${asString(left!)}${asString(right!)}`, +}); + +defineIntrinsic({ + name: "str.codeAt", + doc: "Code point at an ASCII position. The index must be proven in range.", + signature: (args) => { + expectArity("str.codeAt", args, 2); + const text = expectStringClass("str.codeAt", args, 0, "ascii"); + proveIndex("str.codeAt", text, args[1]!); + return text.cls === "digits" ? tInt(0x30n, 0x39n) : tInt(0n, 0x7fn); + }, + evaluate: ([text, index]) => BigInt(codePointsOf(asString(text!))[Number(asBigInt(index!))]!), +}); + +defineIntrinsic({ + name: "str.charAt", + doc: "The one-scalar string at an ASCII position. The index must be proven in range.", + signature: (args) => { + expectArity("str.charAt", args, 2); + const text = expectStringClass("str.charAt", args, 0, "ascii"); + proveIndex("str.charAt", text, args[1]!); + return tString(text.cls, 1, 1); + }, + evaluate: ([text, index]) => asString(text!)[Number(asBigInt(index!))]!, +}); + +defineIntrinsic({ + name: "str.codeAtOpt", + doc: "Code point at an ASCII position, or `none` when the index is outside. The checked form of str.codeAt.", + signature: (args) => { + expectArity("str.codeAtOpt", args, 2); + const text = expectStringClass("str.codeAtOpt", args, 0, "ascii"); + expectKind("str.codeAtOpt", args, 1, "Int"); + return tOption(text.cls === "digits" ? tInt(0x30n, 0x39n) : tInt(0n, 0x7fn)); + }, + evaluate: ([text, index]) => { + const value = asString(text!); + const position = Number(asBigInt(index!)); + return position < 0 || position >= value.length ? NONE : some(BigInt(value.codePointAt(position)!)); + }, +}); + +defineIntrinsic({ + name: "str.charAtOpt", + doc: "The one-scalar string at an ASCII position, or `none` when the index is outside. The checked form of str.charAt.", + signature: (args) => { + expectArity("str.charAtOpt", args, 2); + const text = expectStringClass("str.charAtOpt", args, 0, "ascii"); + expectKind("str.charAtOpt", args, 1, "Int"); + return tOption(tString(text.cls, 1, 1)); + }, + evaluate: ([text, index]) => { + const value = asString(text!); + const position = Number(asBigInt(index!)); + return position < 0 || position >= value.length ? NONE : some(value[position]!); + }, +}); + +defineIntrinsic({ + name: "str.slice", + doc: "A slice of an ASCII string, clamped to its length, from inclusive to exclusive.", + signature: (args) => { + expectArity("str.slice", args, 3); + const text = expectStringClass("str.slice", args, 0, "ascii"); + const from = expectKind("str.slice", args, 1, "Int"); + const to = expectKind("str.slice", args, 2, "Int"); + if (from.lo < 0n || to.lo < 0n) { + throw new SignatureError("str.slice: negative bounds are outside the subset", "clamp the bounds first"); + } + const min = Math.max(0, Number(to.lo - from.hi)); + const max = Math.min(Number(to.hi - from.lo), text.max); + return tString(text.cls, Math.max(0, Math.min(min, max)), Math.max(0, max)); + }, + evaluate: ([text, from, to]) => + fromCodePoints( + codePointsOf(asString(text!)).slice(Number(asBigInt(from!)), Number(asBigInt(to!))), + ), +}); + +defineIntrinsic({ + name: "str.indexOf", + doc: "Scalar index of the first occurrence of `needle`, or -1.", + signature: (args) => { + expectArity("str.indexOf", args, 2); + const text = expectKind("str.indexOf", args, 0, "String"); + expectKind("str.indexOf", args, 1, "String"); + return tInt(-1n, BigInt(Math.max(0, text.max - 1))); + }, + evaluate: ([text, needle]) => { + const haystack = codePointsOf(asString(text!)); + const target = codePointsOf(asString(needle!)); + for (let index = 0; index + target.length <= haystack.length; index++) { + if (target.every((point, offset) => haystack[index + offset] === point)) return BigInt(index); + } + return -1n; + }, +}); + +defineIntrinsic({ + name: "str.contains", + doc: "Whether `needle` occurs in the string.", + signature: (args) => { + expectArity("str.contains", args, 2); + expectKind("str.contains", args, 0, "String"); + expectKind("str.contains", args, 1, "String"); + return { kind: "Bool" }; + }, + evaluate: ([text, needle]) => asString(text!).includes(asString(needle!)), +}); + +defineIntrinsic({ + name: "str.startsWith", + doc: "Whether the string starts with `prefix`.", + signature: (args) => { + expectArity("str.startsWith", args, 2); + expectKind("str.startsWith", args, 0, "String"); + expectKind("str.startsWith", args, 1, "String"); + return { kind: "Bool" }; + }, + evaluate: ([text, prefix]) => asString(text!).startsWith(asString(prefix!)), +}); + +defineIntrinsic({ + name: "str.endsWith", + doc: "Whether the string ends with `suffix`.", + signature: (args) => { + expectArity("str.endsWith", args, 2); + expectKind("str.endsWith", args, 0, "String"); + expectKind("str.endsWith", args, 1, "String"); + return { kind: "Bool" }; + }, + evaluate: ([text, suffix]) => asString(text!).endsWith(asString(suffix!)), +}); + +defineIntrinsic({ + name: "str.repeat", + doc: "The string repeated `count` times.", + signature: (args) => { + expectArity("str.repeat", args, 2); + const text = expectKind("str.repeat", args, 0, "String"); + const count = expectKind("str.repeat", args, 1, "Int"); + if (count.lo < 0n) throw new SignatureError("str.repeat: the count may be negative"); + return tString( + text.cls, + Math.min(text.min * Number(count.lo), MAX_COLLECTION_LENGTH), + Math.min(text.max * Number(count.hi), MAX_COLLECTION_LENGTH), + ); + }, + evaluate: ([text, count]) => asString(text!).repeat(Number(asBigInt(count!))), +}); + +defineIntrinsic({ + name: "str.padStart", + doc: "Left pads with `pad` (one scalar) until the string has `length` scalars.", + signature: (args) => { + expectArity("str.padStart", args, 3); + const text = expectKind("str.padStart", args, 0, "String"); + const length = expectKind("str.padStart", args, 1, "Int"); + const pad = expectKind("str.padStart", args, 2, "String"); + if (pad.min !== 1 || pad.max !== 1) { + throw new SignatureError( + "str.padStart: the padding must be exactly one scalar", + "pass a one character literal; repeated multi-scalar padding truncates differently across targets", + ); + } + return tString( + weakerClass(text.cls, pad.cls), + Math.min(Math.max(text.min, Number(length.lo)), MAX_COLLECTION_LENGTH), + Math.min(Math.max(text.max, Number(length.hi)), MAX_COLLECTION_LENGTH), + ); + }, + evaluate: ([text, length, pad]) => { + const points = codePointsOf(asString(text!)); + const target = Number(asBigInt(length!)); + const missing = Math.max(0, target - points.length); + return `${asString(pad!).repeat(missing)}${asString(text!)}`; + }, +}); + +/** The 25 code points JavaScript's `String#trim` removes, specified explicitly. */ +const JS_WHITESPACE = new Set([ + 0x09, 0x0a, 0x0b, 0x0c, 0x0d, 0x20, 0xa0, 0x1680, 0x2000, 0x2001, 0x2002, 0x2003, 0x2004, 0x2005, + 0x2006, 0x2007, 0x2008, 0x2009, 0x200a, 0x2028, 0x2029, 0x202f, 0x205f, 0x3000, 0xfeff, +]); + +export const TRIM_CODE_POINTS: readonly number[] = [...JS_WHITESPACE].sort((a, b) => a - b); + +defineIntrinsic({ + name: "str.trim", + doc: "Removes leading and trailing whitespace, using the 25 code points JavaScript trims.", + signature: (args) => { + expectArity("str.trim", args, 1); + const text = expectKind("str.trim", args, 0, "String"); + return tString(text.cls, 0, text.max); + }, + evaluate: ([text]) => { + const points = codePointsOf(asString(text!)); + let start = 0; + let end = points.length; + while (start < end && JS_WHITESPACE.has(points[start]!)) start++; + while (end > start && JS_WHITESPACE.has(points[end - 1]!)) end--; + return fromCodePoints(points.slice(start, end)); + }, +}); + +defineIntrinsic({ + name: "str.asciiUpper", + doc: "ASCII-only upper casing: a-z map to A-Z, every other scalar is left alone. Defined on any string; a proven-ASCII argument unlocks the host's own case mapping.", + signature: (args) => { + expectArity("str.asciiUpper", args, 1); + const text = expectKind("str.asciiUpper", args, 0, "String"); + return tString(text.cls, text.min, text.max); + }, + evaluate: ([text]) => asString(text!).replaceAll(/[a-z]/g, (char) => char.toUpperCase()), +}); + +defineIntrinsic({ + name: "str.asciiLower", + doc: "ASCII-only lower casing: A-Z map to a-z, every other scalar is left alone. Defined on any string; a proven-ASCII argument unlocks the host's own case mapping.", + signature: (args) => { + expectArity("str.asciiLower", args, 1); + const text = expectKind("str.asciiLower", args, 0, "String"); + return tString(text.cls, text.min, text.max); + }, + evaluate: ([text]) => asString(text!).replaceAll(/[A-Z]/g, (char) => char.toLowerCase()), +}); + +defineIntrinsic({ + name: "str.compare", + doc: "Scalar-order comparison: -1, 0 or 1. Never the host's collation.", + signature: (args) => { + expectArity("str.compare", args, 2); + expectKind("str.compare", args, 0, "String"); + expectKind("str.compare", args, 1, "String"); + return tInt(-1n, 1n); + }, + evaluate: ([left, right]) => BigInt(compareScalars(asString(left!), asString(right!))), +}); + +defineIntrinsic({ + name: "str.asAscii", + doc: "Checked conversion: `some` when every scalar is below 0x80.", + signature: (args) => { + expectArity("str.asAscii", args, 1); + const text = expectKind("str.asAscii", args, 0, "String"); + return tOption(tString("ascii", text.min, text.max)); + }, + evaluate: ([text]) => { + const value = asString(text!); + return codePointsOf(value).every((point) => point < 0x80) ? some(value) : NONE; + }, +}); + +defineIntrinsic({ + name: "str.asDigits", + doc: "Checked conversion: `some` when every scalar is an ASCII digit.", + signature: (args) => { + expectArity("str.asDigits", args, 1); + const text = expectKind("str.asDigits", args, 0, "String"); + return tOption(tString("digits", text.min, text.max)); + }, + evaluate: ([text]) => { + const value = asString(text!); + const points = codePointsOf(value); + return points.length > 0 && points.every((point) => point >= 0x30 && point <= 0x39) + ? some(value) + : NONE; + }, +}); + +defineIntrinsic({ + name: "str.split", + doc: "Splits on a one-scalar ASCII separator.", + signature: (args) => { + expectArity("str.split", args, 2); + const text = expectKind("str.split", args, 0, "String"); + const separator = expectStringClass("str.split", args, 1, "ascii"); + if (separator.min !== 1 || separator.max !== 1) { + throw new SignatureError("str.split: the separator must be exactly one scalar"); + } + return tList(tString(text.cls, 0, text.max), 1, Math.max(1, text.max + 1)); + }, + evaluate: ([text, separator]) => asString(text!).split(asString(separator!)), +}); + +defineIntrinsic({ + name: "str.join", + doc: "Joins a list of strings with a separator.", + signature: (args) => { + expectArity("str.join", args, 2); + const list = expectKind("str.join", args, 0, "List"); + const separator = expectKind("str.join", args, 1, "String"); + if (list.elem.kind !== "String") throw new SignatureError("str.join: expects a list of strings"); + return tString( + weakerClass(list.elem.cls, separator.cls), + 0, + Math.min(list.max * (list.elem.max + separator.max), MAX_COLLECTION_LENGTH), + ); + }, + evaluate: ([list, separator]) => + asList(list!) + .map((item) => asString(item)) + .join(asString(separator!)), +}); + +defineIntrinsic({ + name: "str.fromInt", + doc: "Decimal representation of an integer, with a leading '-' when negative.", + signature: (args) => { + expectArity("str.fromInt", args, 1); + const value = expectKind("str.fromInt", args, 0, "Int"); + const digits = Math.max(value.lo.toString().length, value.hi.toString().length); + return tString(value.lo >= 0n ? "digits" : "ascii", 1, digits); + }, + evaluate: ([value]) => asBigInt(value!).toString(), +}); + +defineIntrinsic({ + name: "str.parseInt", + doc: "Parses an unsigned decimal integer; `none` when the string is empty or not all digits.", + signature: (args) => { + expectArity("str.parseInt", args, 1); + const text = expectKind("str.parseInt", args, 0, "String"); + const bound = 10n ** BigInt(Math.min(text.max, 18)) - 1n; + return tOption(tInt(0n, bound)); + }, + evaluate: ([text]) => { + const value = asString(text!); + if (value.length === 0 || value.length > 18) return NONE; + return /^[0-9]+$/.test(value) ? some(BigInt(value)) : NONE; + }, +}); diff --git a/engine/src/link/link.ts b/engine/src/link/link.ts new file mode 100644 index 000000000..31f7ea3bf --- /dev/null +++ b/engine/src/link/link.ts @@ -0,0 +1,264 @@ +/** + * Linking: pruning and identical code folding. + * + * Specializing a helper per call site is what lets a refinement cross a function boundary, but + * several specializations often compile to exactly the same code — only the *types* differed, and + * types are proofs, not runtime structure. Folding them back together keeps generated code the + * size a human would have written, and is sound for exactly that reason: each specialization was + * proven safe on its own argument types, and the surviving function's parameter types are the + * join of theirs, which is what a backend reads to pick a representation. + */ + +import type { CExpr, CFunc, CProgram, CStmt } from "../core/ir.ts"; +import { dependencyClosure } from "../analysis/capabilities.ts"; +import { join, typeToString } from "../types.ts"; +import type { SemType } from "../types.ts"; + +export function link(program: CProgram): CProgram { + let current = prune(program); + for (let pass = 0; pass < 8; pass++) { + const folded = foldIdentical(current); + if (folded === current) break; + current = prune(folded); + } + return current; +} + +/** + * Drops everything the entry points cannot reach. + * + * The engine's own standard library is kept regardless: a portable lowering may call one of its + * functions, and which lowering a target selects is not known here (and must not be: nothing + * before the backends may branch on a target). The generator only ever emits the functions its + * own closure reaches, so an unused standard library function never reaches a file. + */ +export function prune(program: CProgram): CProgram { + const standardLibrary = [...program.functions.keys()].filter((name) => name.startsWith("std/")); + const reachable = new Set(dependencyClosure(program, [...program.entryPoints, ...standardLibrary])); + if (reachable.size === program.functions.size) return program; + const functions = new Map(); + for (const [name, fn] of program.functions) { + if (reachable.has(name)) functions.set(name, fn); + } + return { ...program, functions }; +} + +function foldIdentical(program: CProgram): CProgram { + const byKey = new Map(); + const rename = new Map(); + // Callees first, so a fold propagates up into the callers' keys on this same pass. + for (const name of dependencyClosure(program, program.entryPoints)) { + const fn = program.functions.get(name); + if (fn === undefined || program.entryPoints.includes(name)) continue; + const key = structuralKey(fn, rename); + const existing = byKey.get(key); + if (existing === undefined) { + byKey.set(key, name); + continue; + } + rename.set(name, existing); + } + if (rename.size === 0) return program; + + const functions = new Map(); + for (const [name, fn] of program.functions) { + if (rename.has(name)) continue; + const merged = mergeParams(fn, program, rename, name); + functions.set(name, { ...merged, body: rewriteCalls(merged.body, rename), calls: merged.calls.map((callee) => rename.get(callee) ?? callee) }); + } + return { ...program, functions }; +} + +/** The surviving function must accept every call site of the ones folded into it. */ +function mergeParams( + fn: CFunc, + program: CProgram, + rename: ReadonlyMap, + survivor: string, +): CFunc { + const merged = [...fn.params]; + for (const [from, to] of rename) { + if (to !== survivor) continue; + const other = program.functions.get(from); + if (other === undefined) continue; + other.params.forEach((param, index) => { + const existing = merged[index]; + if (existing !== undefined) merged[index] = { ...existing, type: join(existing.type, param.type) }; + }); + } + return { ...fn, params: merged }; +} + +function rewriteCalls(body: readonly CStmt[], rename: ReadonlyMap): CStmt[] { + const expr = (node: CExpr): CExpr => { + switch (node.kind) { + case "call": + return { ...node, fn: rename.get(node.fn) ?? node.fn, args: node.args.map(expr) }; + case "op": + return { ...node, args: node.args.map(expr) }; + case "record": + return { ...node, fields: node.fields.map((field) => ({ ...field, value: expr(field.value) })) }; + case "list": + return { ...node, items: node.items.map(expr) }; + case "field": + return { ...node, target: expr(node.target) }; + case "some": + return { ...node, inner: expr(node.inner) }; + case "cond": + return { ...node, test: expr(node.test), then: expr(node.then), otherwise: expr(node.otherwise) }; + case "and": + case "or": + return { ...node, left: expr(node.left), right: expr(node.right) }; + case "not": + return { ...node, operand: expr(node.operand) }; + case "lambda": + return { ...node, body: rewriteCalls(node.body, rename) }; + default: + return node; + } + }; + return body.map((statement): CStmt => { + switch (statement.kind) { + case "let": + return { ...statement, init: expr(statement.init) }; + case "assign": + return { ...statement, value: expr(statement.value) }; + case "setIndex": + return { ...statement, index: expr(statement.index), value: expr(statement.value) }; + case "push": + return { ...statement, value: expr(statement.value) }; + case "if": + return { + ...statement, + test: expr(statement.test), + then: rewriteCalls(statement.then, rename), + otherwise: rewriteCalls(statement.otherwise, rename), + }; + case "switch": + return { + ...statement, + subject: expr(statement.subject), + cases: statement.cases.map((entry) => ({ ...entry, body: rewriteCalls(entry.body, rename) })), + otherwise: statement.otherwise === undefined ? undefined : rewriteCalls(statement.otherwise, rename), + }; + case "forRange": + return { + ...statement, + from: expr(statement.from), + to: expr(statement.to), + body: rewriteCalls(statement.body, rename), + }; + case "forEach": + return { ...statement, iterable: expr(statement.iterable), body: rewriteCalls(statement.body, rename) }; + case "return": + return statement.value === undefined ? statement : { ...statement, value: expr(statement.value) }; + case "fail": + return { ...statement, args: statement.args.map(expr) }; + case "expr": + return { ...statement, expr: expr(statement.expr) }; + default: + return statement; + } + }); +} + +/** + * A key that ignores everything a backend does not print: the refinements in types, and the + * specialization suffix in a callee's name once it has been folded. + */ +function structuralKey(fn: CFunc, rename: ReadonlyMap): string { + const parts: string[] = [ + fn.params.map((param) => `${param.name}:${shapeOf(param.type)}`).join(","), + shapeOf(fn.ret), + fn.effects.fail.join("|"), + String(fn.usesEnv), + ]; + const expr = (node: CExpr): string => { + switch (node.kind) { + case "lit": + return `lit(${String(node.value)})`; + case "local": + return `loc(${node.name})`; + case "none": + return "none"; + case "some": + return `some(${expr(node.inner)})`; + case "record": + return `rec(${node.typeName},${node.fields.map((field) => `${field.name}=${expr(field.value)}`).join(",")})`; + case "field": + return `fld(${expr(node.target)},${node.name})`; + case "list": + return `lst(${node.items.map(expr).join(",")})`; + case "call": + return `call(${rename.get(node.fn) ?? node.fn},${node.args.map(expr).join(",")})`; + case "op": + return `op(${node.op}${node.regex === undefined ? "" : `/${node.regex.source}/`},${node.args.map(expr).join(",")})`; + case "lambda": + return `lam(${node.params.map((param) => param.name).join(",")},${node.body.map(statement).join(";")})`; + case "cond": + return `cond(${expr(node.test)},${expr(node.then)},${expr(node.otherwise)})`; + case "and": + return `and(${expr(node.left)},${expr(node.right)})`; + case "or": + return `or(${expr(node.left)},${expr(node.right)})`; + case "not": + return `not(${expr(node.operand)})`; + default: { + const exhaustive: never = node; + return exhaustive; + } + } + }; + const statement = (node: CStmt): string => { + switch (node.kind) { + case "let": + return `let ${node.name}=${expr(node.init)}`; + case "assign": + return `${node.name}=${expr(node.value)}`; + case "setIndex": + return `${node.name}[${expr(node.index)}]=${expr(node.value)}`; + case "push": + return `push ${node.name},${expr(node.value)}`; + case "if": + return `if(${expr(node.test)}){${node.then.map(statement).join(";")}}else{${node.otherwise.map(statement).join(";")}}`; + case "switch": + return `switch(${expr(node.subject)}){${node.cases.map((entry) => `${entry.values.join("|")}:${entry.body.map(statement).join(";")}`).join("|")}}${node.otherwise === undefined ? "" : `default:${node.otherwise.map(statement).join(";")}`}`; + case "forRange": + return `for(${node.name},${expr(node.from)},${expr(node.to)},${node.inclusive},${node.step}){${node.body.map(statement).join(";")}}`; + case "forEach": + return `each(${node.name},${expr(node.iterable)}){${node.body.map(statement).join(";")}}`; + case "return": + return `ret(${node.value === undefined ? "" : expr(node.value)})`; + case "fail": + return `fail(${node.errorClass},${node.args.map(expr).join(",")})`; + case "break": + return "break"; + case "continue": + return "continue"; + case "expr": + return `do(${expr(node.expr)})`; + default: { + const exhaustive: never = node; + return exhaustive; + } + } + }; + parts.push(fn.body.map(statement).join(";")); + return parts.join("#"); +} + +/** The part of a type a backend can see: the shape, never the proof. */ +function shapeOf(type: SemType): string { + switch (type.kind) { + case "Int": + return "Int"; + case "String": + return "String"; + case "List": + return `List<${shapeOf(type.elem)}>`; + case "Option": + return `Option<${shapeOf(type.inner)}>`; + default: + return typeToString(type); + } +} diff --git a/engine/src/optimize/inline.ts b/engine/src/optimize/inline.ts new file mode 100644 index 000000000..3eee424d9 --- /dev/null +++ b/engine/src/optimize/inline.ts @@ -0,0 +1,1108 @@ +/** + * Call-site inlining, target scoped. + * + * A small function called in a loop is free once a JIT or an optimizing compiler decides to inline + * it, and expensive when it is not: CPython pays a full frame per call, V8 usually elides the + * closure once a call site is hot, and rustc inlines across crate boundaries only when LTO is on. + * That is a per-target cost decision, the same as every other choice this engine makes, so this + * pass runs once per target — from `generate` (`backend/generate.ts`), with that target's own + * budget — rather than once for the whole program the way `optimize.ts`'s passes do. + * + * Every call inlined here is provably a straight substitution: the callee's parameters are bound + * once each (never re-evaluated, whatever the argument expression costs), every one of its locals + * is renamed to a fresh name unique to this occurrence (so two call sites inlined into the same + * function never collide), and an early `return` is turned into an assignment to a synthetic + * `Option` result plus a `break` where it is inside a loop — the same "was this the answer yet" + * question every target already has a native lowering for (`opt.isNone`, `opt.unwrap`), so nothing + * new has to be taught to any backend. A callee that fails, reaches `Http`, calls itself + * (directly or through another inlined callee) or holds a lambda is left as an ordinary call: none + * of those are unsound to inline in principle, they are just outside what this pass proves safe in + * the time it has to prove it. + * + * Inlining also has a price, and for one target that price is the one the package is judged on. + * The generated TypeScript is shipped to a browser and the npm package is tree-shakeable, which + * ADR 0012 records as a requirement rather than a preference; a copy of a callee's body is bytes + * over the wire, once per copy. So the budget is not only "how big is the callee" but "how much + * code does this add", which is what `maxGrowthStatements` bounds. The two questions have + * different answers: a callee spliced into its only remaining call site takes its own definition + * with it and adds nothing at all, however big it is, while a three-statement helper called nine + * times adds eight copies of itself. `engine/scripts/size.ts` measures the result the way a + * consumer's bundler would, and `core/bench` measures what it bought. + */ + +import type { CExpr, CFunc, CProgram, CStmt } from "../core/ir.ts"; +import type { SemType } from "../types.ts"; +import { tOption } from "../types.ts"; +import { dependencyClosure } from "../analysis/capabilities.ts"; + +export type InlineBudget = { + /** The callee's own statement count (recursive), above which it is left as a call. 0 disables the pass. */ + readonly maxStatements: number; + /** How many times to re-scan for a newly exposed call (e.g. inlining `f` reveals `f`'s own call to `g`). */ + readonly rounds?: number; + /** + * The most Core nodes any one inline may *add* to the program. Absent means unbounded, which + * is what a target compiled ahead of time wants: its cost is frames at run time and nobody + * downloads its source. A target whose output is shipped over the wire sets it, and what it + * says is "duplicate only what is small". + * + * Nodes rather than statements, because the unit has to hold for both shapes this pass emits. + * A one-expression helper is a single statement whether it reads `a + b` or spans half a + * screen, and counting it as one would let the second be copied to nine call sites for the + * price of the first. + * + * An inline that leaves the callee with no call sites at all adds almost nothing however big + * the callee is: its definition is dropped from the dependency closure (`backend/lower.ts`'s + * `closure`) the moment nothing reaches it, so the code moved rather than multiplied, and only + * whatever the splice added on top is counted. `0` therefore means exactly "take the inlines + * that pay for themselves, and no others" — not "inline nothing". + * + * A cap per inline rather than a pool for the whole pass, so the answer does not depend on + * which call site the walk happened to reach first. + */ + readonly maxDuplicatedNodes?: number; +}; + +export function inlineCalls(program: CProgram, budget: InlineBudget): CProgram { + if (budget.maxStatements <= 0) return program; + + const recursive = recursiveFunctions(program); + let functions = program.functions; + const rounds = budget.rounds ?? 3; + // One counter for the whole pass, not one per function or per round: a fresh name only has to + // be unique within the function it lands in, but scoping the counter that narrowly means two + // different rounds processing the same function can hand out the same name to two different + // splices — the second round sees the first round's own hoisted `let`s as ordinary statements, + // not as already-claimed names. A single counter for every splice this call makes never repeats. + const uid = { n: 0 }; + + for (let round = 0; round < rounds; round++) { + let changedThisRound = false; + const next = new Map(functions); + // Recounted every round, because an inline is itself a call site moving: splicing `f` into + // its caller copies every call `f` made, and a round that ran before this one may already + // have emptied a callee the next one would otherwise still think is shared. + const sites = callSites({ ...program, functions }); + for (const [name, fn] of functions) { + const body = inlineBody(fn.body, { program, budget, recursive, uid, callerName: name, sites }); + if (body !== fn.body) { + changedThisRound = true; + next.set(name, { ...fn, body, calls: collectCalls(body) }); + } + } + functions = next; + if (!changedThisRound) break; + } + + return { ...program, functions }; +} + +/** + * How many times each function is called, counted over the functions that will actually be + * emitted — the dependency closure of the entry points, the same set `backend/lower.ts` lowers. + * A call from a function nobody reaches is not a copy anyone downloads, and counting it would + * keep a callee looking shared when its only real caller is about to absorb it. + */ +function callSites(program: CProgram): Map { + const counts = new Map(); + for (const name of dependencyClosure(program, program.entryPoints)) { + const fn = program.functions.get(name); + if (fn === undefined) continue; + countCalls(fn.body, counts); + } + return counts; +} + +type Ctx = { + readonly program: CProgram; + readonly budget: InlineBudget; + readonly recursive: ReadonlySet; + readonly uid: { n: number }; + readonly callerName: string; + /** Call sites per callee at the start of this round, decremented as this round consumes them. */ + readonly sites: Map; +}; + +/* ------------------------------------------------------------------ * + * Eligibility + * ------------------------------------------------------------------ */ + +function statementCount(body: readonly CStmt[]): number { + let total = 0; + for (const statement of body) { + total += 1; + switch (statement.kind) { + case "if": + total += statementCount(statement.then) + statementCount(statement.otherwise); + break; + case "switch": + total += statement.cases.reduce((sum, entry) => sum + statementCount(entry.body), 0); + total += statement.otherwise === undefined ? 0 : statementCount(statement.otherwise); + break; + case "forRange": + case "forEach": + total += statementCount(statement.body); + break; + default: + break; + } + } + return total; +} + +function containsLambda(body: readonly CStmt[]): boolean { + const inExpr = (expr: CExpr): boolean => { + switch (expr.kind) { + case "lambda": + return true; + case "some": + return inExpr(expr.inner); + case "record": + return expr.fields.some((field) => inExpr(field.value)); + case "field": + return inExpr(expr.target); + case "list": + return expr.items.some(inExpr); + case "call": + case "op": + return expr.args.some(inExpr); + case "cond": + return inExpr(expr.test) || inExpr(expr.then) || inExpr(expr.otherwise); + case "and": + case "or": + return inExpr(expr.left) || inExpr(expr.right); + case "not": + return inExpr(expr.operand); + default: + return false; + } + }; + const inStmt = (statement: CStmt): boolean => { + switch (statement.kind) { + case "let": + return inExpr(statement.init); + case "assign": + return inExpr(statement.value); + case "setIndex": + return inExpr(statement.index) || inExpr(statement.value); + case "push": + return inExpr(statement.value); + case "if": + return inExpr(statement.test) || statement.then.some(inStmt) || statement.otherwise.some(inStmt); + case "switch": + return ( + inExpr(statement.subject) || + statement.cases.some((entry) => entry.body.some(inStmt)) || + (statement.otherwise?.some(inStmt) ?? false) + ); + case "forRange": + return inExpr(statement.from) || inExpr(statement.to) || statement.body.some(inStmt); + case "forEach": + return inExpr(statement.iterable) || statement.body.some(inStmt); + case "return": + return statement.value !== undefined && inExpr(statement.value); + case "fail": + return statement.args.some(inExpr); + case "expr": + return inExpr(statement.expr); + default: + return false; + } + }; + return body.some(inStmt); +} + +/** Every function that calls itself, directly or through another function it calls. */ +function recursiveFunctions(program: CProgram): ReadonlySet { + const recursive = new Set(); + for (const [name] of program.functions) { + const seen = new Set(); + const stack = [name]; + while (stack.length > 0) { + const current = stack.pop()!; + const fn = program.functions.get(current); + if (fn === undefined) continue; + for (const callee of fn.calls) { + if (callee === name) { + recursive.add(name); + stack.length = 0; + break; + } + if (!seen.has(callee)) { + seen.add(callee); + stack.push(callee); + } + } + } + } + return recursive; +} + +function eligible(callee: CFunc, ctx: Ctx): boolean { + if (callee.effects.fail.length > 0) return false; + if (callee.effects.http) return false; + if (callee.ret.kind === "Void") return false; + if (ctx.recursive.has(callee.name)) return false; + if (containsLambda(callee.body)) return false; + return statementCount(callee.body) <= ctx.budget.maxStatements; +} + +/** + * Whether the budget covers a splice of `spliced` nodes in place of a call to `callee`, and if so, + * spends it. + * + * The price is what is actually emitted, not what the callee's source looks like: a body with an + * early `return` grows by the sentinel protocol `flattenSeq` introduces, and a body that is one + * expression grows by that expression. Against that, an inline that takes the callee's last call + * site refunds the whole definition, which falls out of the dependency closure + * (`backend/lower.ts`'s `closure`) the moment nothing reaches it — so a sole-call-site helper is + * free exactly when the copy is no bigger than the definition it replaces, and not by assumption. + * + * An entry point is never refunded: the closure keeps it whether or not anything calls it. + */ +function affordable(callee: CFunc, spliced: number, ctx: Ctx): boolean { + const cap = ctx.budget.maxDuplicatedNodes; + const remaining = ctx.sites.get(callee.name) ?? 0; + const lastCall = remaining <= 1 && !ctx.program.entryPoints.includes(callee.name); + const growth = spliced - (lastCall ? nodeCount(callee.body) : 0); + if (cap !== undefined && growth > cap) return false; + ctx.sites.set(callee.name, Math.max(0, remaining - 1)); + return true; +} + +/** Nodes in a statement list: every statement, and every expression node it holds. */ +function nodeCount(body: readonly CStmt[]): number { + let total = 0; + for (const statement of body) { + total += 1; + switch (statement.kind) { + case "let": + total += exprNodes(statement.init); + break; + case "assign": + total += exprNodes(statement.value); + break; + case "setIndex": + total += exprNodes(statement.index) + exprNodes(statement.value); + break; + case "push": + total += exprNodes(statement.value); + break; + case "if": + total += exprNodes(statement.test) + nodeCount(statement.then) + nodeCount(statement.otherwise); + break; + case "switch": + total += exprNodes(statement.subject); + total += statement.cases.reduce((sum, entry) => sum + nodeCount(entry.body), 0); + total += statement.otherwise === undefined ? 0 : nodeCount(statement.otherwise); + break; + case "forRange": + total += exprNodes(statement.from) + exprNodes(statement.to) + nodeCount(statement.body); + break; + case "forEach": + total += exprNodes(statement.iterable) + nodeCount(statement.body); + break; + case "return": + total += statement.value === undefined ? 0 : exprNodes(statement.value); + break; + case "fail": + total += statement.args.reduce((sum, arg) => sum + exprNodes(arg), 0); + break; + case "expr": + total += exprNodes(statement.expr); + break; + default: + break; + } + } + return total; +} + +function exprNodes(expr: CExpr): number { + switch (expr.kind) { + case "lit": + case "local": + case "none": + return 1; + case "some": + return 1 + exprNodes(expr.inner); + case "record": + return 1 + expr.fields.reduce((sum, field) => sum + exprNodes(field.value), 0); + case "field": + return 1 + exprNodes(expr.target); + case "list": + return 1 + expr.items.reduce((sum, item) => sum + exprNodes(item), 0); + case "call": + case "op": + return 1 + expr.args.reduce((sum, arg) => sum + exprNodes(arg), 0); + case "lambda": + return 1 + nodeCount(expr.body); + case "cond": + return 1 + exprNodes(expr.test) + exprNodes(expr.then) + exprNodes(expr.otherwise); + case "and": + case "or": + return 1 + exprNodes(expr.left) + exprNodes(expr.right); + case "not": + return 1 + exprNodes(expr.operand); + default: { + const exhaustive: never = expr; + return exhaustive; + } + } +} + +/** Call counts, accumulated into `counts`; `collectCalls` answers the set, this one the tally. */ +function countCalls(body: readonly CStmt[], counts: Map): void { + for (const name of collectCallsWithRepeats(body)) counts.set(name, (counts.get(name) ?? 0) + 1); +} + +/* ------------------------------------------------------------------ * + * Renaming: every callee-local gets a fresh name unique to one splice + * ------------------------------------------------------------------ */ + +function collectBoundNames(body: readonly CStmt[], names: Set): void { + for (const statement of body) { + switch (statement.kind) { + case "let": + names.add(statement.name); + break; + case "if": + collectBoundNames(statement.then, names); + collectBoundNames(statement.otherwise, names); + break; + case "switch": + for (const entry of statement.cases) collectBoundNames(entry.body, names); + if (statement.otherwise !== undefined) collectBoundNames(statement.otherwise, names); + break; + case "forRange": + names.add(statement.name); + collectBoundNames(statement.body, names); + break; + case "forEach": + names.add(statement.name); + collectBoundNames(statement.body, names); + break; + default: + break; + } + } +} + +function renameExpr(expr: CExpr, map: ReadonlyMap): CExpr { + switch (expr.kind) { + case "lit": + case "none": + return expr; + case "local": { + const to = map.get(expr.name); + return to === undefined ? expr : { ...expr, name: to }; + } + case "some": + return { ...expr, inner: renameExpr(expr.inner, map) }; + case "record": + return { ...expr, fields: expr.fields.map((field) => ({ ...field, value: renameExpr(field.value, map) })) }; + case "field": + return { ...expr, target: renameExpr(expr.target, map) }; + case "list": + return { ...expr, items: expr.items.map((item) => renameExpr(item, map)) }; + case "call": + case "op": + return { ...expr, args: expr.args.map((arg) => renameExpr(arg, map)) }; + case "lambda": + // Excluded by `containsLambda` before a callee ever reaches this function. + return expr; + case "cond": + return { + ...expr, + test: renameExpr(expr.test, map), + then: renameExpr(expr.then, map), + otherwise: renameExpr(expr.otherwise, map), + }; + case "and": + case "or": + return { ...expr, left: renameExpr(expr.left, map), right: renameExpr(expr.right, map) }; + case "not": + return { ...expr, operand: renameExpr(expr.operand, map) }; + default: { + const exhaustive: never = expr; + return exhaustive; + } + } +} + +function renameStmt(statement: CStmt, map: ReadonlyMap): CStmt { + switch (statement.kind) { + case "let": + return { ...statement, name: map.get(statement.name) ?? statement.name, init: renameExpr(statement.init, map) }; + case "assign": + return { ...statement, name: map.get(statement.name) ?? statement.name, value: renameExpr(statement.value, map) }; + case "setIndex": + return { + ...statement, + name: map.get(statement.name) ?? statement.name, + index: renameExpr(statement.index, map), + value: renameExpr(statement.value, map), + }; + case "push": + return { ...statement, name: map.get(statement.name) ?? statement.name, value: renameExpr(statement.value, map) }; + case "if": + return { + ...statement, + test: renameExpr(statement.test, map), + then: statement.then.map((item) => renameStmt(item, map)), + otherwise: statement.otherwise.map((item) => renameStmt(item, map)), + }; + case "switch": + return { + ...statement, + subject: renameExpr(statement.subject, map), + cases: statement.cases.map((entry) => ({ ...entry, body: entry.body.map((item) => renameStmt(item, map)) })), + otherwise: statement.otherwise?.map((item) => renameStmt(item, map)), + }; + case "forRange": + return { + ...statement, + name: map.get(statement.name) ?? statement.name, + from: renameExpr(statement.from, map), + to: renameExpr(statement.to, map), + body: statement.body.map((item) => renameStmt(item, map)), + }; + case "forEach": + return { + ...statement, + name: map.get(statement.name) ?? statement.name, + iterable: renameExpr(statement.iterable, map), + body: statement.body.map((item) => renameStmt(item, map)), + }; + case "return": + return statement.value === undefined ? statement : { ...statement, value: renameExpr(statement.value, map) }; + case "fail": + return { ...statement, args: statement.args.map((arg) => renameExpr(arg, map)) }; + case "break": + case "continue": + return statement; + case "expr": + return { ...statement, expr: renameExpr(statement.expr, map) }; + default: { + const exhaustive: never = statement; + return exhaustive; + } + } +} + +/** Whether `name` is ever the target of an `assign` inside `body` (not crossing a lambda — excluded already). */ +function isReassigned(body: readonly CStmt[], name: string): boolean { + return body.some((statement): boolean => { + switch (statement.kind) { + case "assign": + return statement.name === name; + case "if": + return isReassigned(statement.then, name) || isReassigned(statement.otherwise, name); + case "switch": + return ( + statement.cases.some((entry) => isReassigned(entry.body, name)) || + (statement.otherwise !== undefined && isReassigned(statement.otherwise, name)) + ); + case "forRange": + case "forEach": + return isReassigned(statement.body, name); + default: + return false; + } + }); +} + +/* ------------------------------------------------------------------ * + * Turning an early `return` into an assignment to a synthetic Option + * ------------------------------------------------------------------ */ + +function stmtsContainReturn(body: readonly CStmt[]): boolean { + return body.some((statement): boolean => { + switch (statement.kind) { + case "return": + return true; + case "if": + return stmtsContainReturn(statement.then) || stmtsContainReturn(statement.otherwise); + case "switch": + return ( + statement.cases.some((entry) => stmtsContainReturn(entry.body)) || + (statement.otherwise !== undefined && stmtsContainReturn(statement.otherwise)) + ); + case "forRange": + case "forEach": + return stmtsContainReturn(statement.body); + default: + return false; + } + }); +} + +function isNoneTest(resultVar: string, optionType: SemType, span: CExpr["span"]): CExpr { + return { kind: "op", op: "opt.isNone", args: [{ kind: "local", name: resultVar, type: optionType, span }], type: { kind: "Bool" }, span }; +} + +/** + * Rewrites a callee body (already renamed to this splice's fresh names) so every `return expr` + * becomes `resultVar = some(expr)`, breaking the nearest loop it is directly inside, and every + * statement that can no longer be reached once `resultVar` is no longer absent is guarded by + * `if (opt.isNone(resultVar)) { … }`. This is the same early-return-to-flag rewrite every compiler + * that desugars `return` out of structured control flow performs; nothing here is Core-specific. + */ +function flattenSeq(body: readonly CStmt[], resultVar: string, retType: SemType, optionType: SemType): CStmt[] { + if (body.length === 0) return []; + const [first, ...rest] = body; + const flatFirst = flattenStmt(first!, resultVar, retType, optionType); + if (rest.length === 0) return flatFirst; + const flatRest = flattenSeq(rest, resultVar, retType, optionType); + if (!stmtContainsReturn(first!)) return [...flatFirst, ...flatRest]; + return [ + ...flatFirst, + { kind: "if", test: isNoneTest(resultVar, optionType, first!.span), then: flatRest, otherwise: [], span: first!.span }, + ]; +} + +function stmtContainsReturn(statement: CStmt): boolean { + switch (statement.kind) { + case "return": + return true; + case "if": + return stmtsContainReturn(statement.then) || stmtsContainReturn(statement.otherwise); + case "switch": + return ( + statement.cases.some((entry) => stmtsContainReturn(entry.body)) || + (statement.otherwise !== undefined && stmtsContainReturn(statement.otherwise)) + ); + case "forRange": + case "forEach": + return stmtsContainReturn(statement.body); + default: + return false; + } +} + +function flattenStmt(statement: CStmt, resultVar: string, retType: SemType, optionType: SemType): CStmt[] { + switch (statement.kind) { + case "return": + // `eligible` requires a non-`Void` return type, so every `return` in an inlinable callee + // carries a value. + return [ + { + kind: "assign", + name: resultVar, + value: { kind: "some", inner: statement.value!, type: optionType, span: statement.span }, + span: statement.span, + }, + ]; + case "if": + return [ + { + ...statement, + then: flattenSeq(statement.then, resultVar, retType, optionType), + otherwise: flattenSeq(statement.otherwise, resultVar, retType, optionType), + }, + ]; + case "switch": + return [ + { + ...statement, + cases: statement.cases.map((entry) => ({ + ...entry, + body: flattenSeq(entry.body, resultVar, retType, optionType), + })), + otherwise: + statement.otherwise === undefined ? undefined : flattenSeq(statement.otherwise, resultVar, retType, optionType), + }, + ]; + case "forRange": + case "forEach": { + let body = flattenSeq(statement.body, resultVar, retType, optionType); + if (stmtsContainReturn(statement.body)) { + // A `return` inside this loop only breaks the loop it is textually inside (an ordinary + // `break`, nothing crosses a loop boundary); an outer loop stops in turn because *it* + // sees this whole `forRange`/`forEach` as "might have set the result" and gets the same + // guard appended below it, one level up, the next time `flattenSeq` walks its siblings. + body = [ + ...body, + { + kind: "if", + test: { kind: "not", operand: isNoneTest(resultVar, optionType, statement.span), type: { kind: "Bool" }, span: statement.span }, + then: [{ kind: "break", span: statement.span }], + otherwise: [], + span: statement.span, + }, + ]; + } + return [{ ...statement, body }]; + } + default: + return [statement]; + } +} + +/* ------------------------------------------------------------------ * + * The walk: find an eligible call, splice its (flattened, renamed) body in + * ------------------------------------------------------------------ */ + +function inlineBody(body: readonly CStmt[], ctx: Ctx): readonly CStmt[] { + const result = body.flatMap((statement) => inlineStmt(statement, ctx)); + return result.length === body.length && result.every((statement, index) => statement === body[index]) ? body : result; +} + +function inlineStmt(statement: CStmt, ctx: Ctx): CStmt[] { + const hoisted: CStmt[] = []; + const at = (expr: CExpr, conditional: boolean): CExpr => inlineExpr(expr, hoisted, ctx, conditional); + let rewritten: CStmt; + switch (statement.kind) { + case "let": + rewritten = { ...statement, init: at(statement.init, false) }; + break; + case "assign": + rewritten = { ...statement, value: at(statement.value, false) }; + break; + case "setIndex": + rewritten = { ...statement, index: at(statement.index, false), value: at(statement.value, false) }; + break; + case "push": + rewritten = { ...statement, value: at(statement.value, false) }; + break; + case "if": + rewritten = { + ...statement, + test: at(statement.test, false), + then: inlineBody(statement.then, ctx), + otherwise: inlineBody(statement.otherwise, ctx), + }; + break; + case "switch": + rewritten = { + ...statement, + subject: at(statement.subject, false), + cases: statement.cases.map((entry) => ({ ...entry, body: inlineBody(entry.body, ctx) })), + otherwise: statement.otherwise === undefined ? undefined : inlineBody(statement.otherwise, ctx), + }; + break; + case "forRange": + rewritten = { ...statement, from: at(statement.from, false), to: at(statement.to, false), body: inlineBody(statement.body, ctx) }; + break; + case "forEach": + rewritten = { ...statement, iterable: at(statement.iterable, false), body: inlineBody(statement.body, ctx) }; + break; + case "return": + rewritten = statement.value === undefined ? statement : { ...statement, value: at(statement.value, false) }; + break; + case "fail": + rewritten = { ...statement, args: statement.args.map((arg) => at(arg, false)) }; + break; + case "break": + case "continue": + rewritten = statement; + break; + case "expr": + rewritten = { ...statement, expr: at(statement.expr, false) }; + break; + default: { + const exhaustive: never = statement; + rewritten = exhaustive; + } + } + return hoisted.length === 0 ? [rewritten] : [...hoisted, rewritten]; +} + +/** + * `conditional` is true inside an expression that is not always evaluated — the right side of + * `&&`/`||`, or either branch of `cond` — where hoisting a call's setup statements ahead of the + * whole expression would run it unconditionally. A call found there is a candidate only for a + * splice that needs no statements; everywhere else (including `cond`'s own `test`, and every + * argument of an ordinary call/op/record/list, which always run) every splice is. + */ +function inlineExpr(expr: CExpr, hoisted: CStmt[], ctx: Ctx, conditional: boolean): CExpr { + switch (expr.kind) { + case "lit": + case "local": + case "none": + return expr; + case "some": + return { ...expr, inner: inlineExpr(expr.inner, hoisted, ctx, conditional) }; + case "record": + return { ...expr, fields: expr.fields.map((field) => ({ ...field, value: inlineExpr(field.value, hoisted, ctx, conditional) })) }; + case "field": + return { ...expr, target: inlineExpr(expr.target, hoisted, ctx, conditional) }; + case "list": + return { ...expr, items: expr.items.map((item) => inlineExpr(item, hoisted, ctx, conditional)) }; + case "op": + return { ...expr, args: expr.args.map((arg) => inlineExpr(arg, hoisted, ctx, conditional)) }; + case "cond": + return { + ...expr, + test: inlineExpr(expr.test, hoisted, ctx, conditional), + then: inlineExpr(expr.then, hoisted, ctx, true), + otherwise: inlineExpr(expr.otherwise, hoisted, ctx, true), + }; + case "and": + case "or": + return { + ...expr, + left: inlineExpr(expr.left, hoisted, ctx, conditional), + right: inlineExpr(expr.right, hoisted, ctx, true), + }; + case "not": + return { ...expr, operand: inlineExpr(expr.operand, hoisted, ctx, conditional) }; + case "lambda": + return expr; + case "call": { + const withArgsDone: CExpr = { ...expr, args: expr.args.map((arg) => inlineExpr(arg, hoisted, ctx, conditional)) }; + // A splice that needs no statements at all is safe here too, and this is where it helps + // most: a one-expression helper called inside a ternary or behind `&&` is exactly the + // call a reader expected to see substituted. + const inlined = tryInline(withArgsDone as Extract, hoisted, ctx, !conditional); + return inlined ?? withArgsDone; + } + default: { + const exhaustive: never = expr; + return exhaustive; + } + } +} + +/** + * `mayHoist` is false where the call sits in an expression that is not always evaluated — the + * right side of `&&`/`||`, either branch of a ternary — and lifting setup statements ahead of the + * whole expression would run them unconditionally. Only a splice that needs no statements at all + * is taken there. + */ +function tryInline( + call: Extract, + hoisted: CStmt[], + ctx: Ctx, + mayHoist: boolean, +): CExpr | undefined { + const callee = ctx.program.functions.get(call.fn); + if (callee === undefined || !eligible(callee, ctx)) return undefined; + + const fresh = (base: string): string => { + ctx.uid.n += 1; + return `__inl${ctx.uid.n}_${base}`; + }; + + const asExpression = tryInlineExpression(call, callee, fresh); + if (asExpression !== undefined) { + if (!mayHoist && asExpression.hoisted.length > 0) return undefined; + const spliced = nodeCount(asExpression.hoisted) + exprNodes(asExpression.value); + if (!affordable(callee, spliced, ctx)) return undefined; + hoisted.push(...asExpression.hoisted); + return asExpression.value; + } + if (!mayHoist) return undefined; + + const renameMap = new Map(); + const bound = new Set(); + collectBoundNames(callee.body, bound); + for (const name of bound) renameMap.set(name, fresh(name)); + for (const param of callee.params) renameMap.set(param.name, fresh(param.name)); + + const splice: CStmt[] = []; + // A parameter the call passes a literal or a local for is substituted rather than bound: the + // `let` would only be an alias, and `const _inl117_year = year;` above the body is scaffolding + // no author would leave in. Anything else — a call, an arithmetic expression, anything that + // may have an effect or cost something — is bound once, in argument order, so the body reading + // it twice cannot evaluate it twice. A parameter the callee assigns to needs its own storage + // whatever the argument was. + const substitution = new Map(); + for (let index = 0; index < callee.params.length; index++) { + const param = callee.params[index]!; + const arg = call.args[index]!; + const name = renameMap.get(param.name)!; + const reassigned = isReassigned(callee.body, param.name); + if (!reassigned && (arg.kind === "lit" || arg.kind === "local")) { + substitution.set(name, arg); + continue; + } + splice.push({ kind: "let", name, mutable: reassigned, init: arg, type: param.type, span: call.span }); + } + + const resultVar = fresh("result"); + const renamedBody = callee.body.map((statement) => renameStmt(statement, renameMap)); + const straightLine = trailingReturnOnly(renamedBody); + if (straightLine !== undefined) { + // The common shape, and the one worth not paying for: every statement runs, then the last + // one returns. There is nothing for a sentinel to answer, so the body is spliced as it + // stands and the returned expression is bound to one `const`. + splice.push(...straightLine.before); + splice.push({ kind: "let", name: resultVar, mutable: false, init: straightLine.value, type: callee.ret, span: call.span }); + } else { + const optionType = tOption(callee.ret); + splice.push({ + kind: "let", + name: resultVar, + mutable: true, + init: { kind: "none", type: optionType, span: call.span }, + type: optionType, + span: call.span, + }); + splice.push(...flattenSeq(renamedBody, resultVar, callee.ret, optionType)); + } + + const substituted = substitution.size === 0 ? splice : splice.map((statement) => substituteStmt(statement, substitution)); + if (!affordable(callee, nodeCount(substituted), ctx)) return undefined; + hoisted.push(...substituted); + + if (straightLine !== undefined) { + return { kind: "local", name: resultVar, type: callee.ret, span: call.span }; + } + const optionType = tOption(callee.ret); + return { kind: "op", op: "opt.unwrap", args: [{ kind: "local", name: resultVar, type: optionType, span: call.span }], type: call.type, span: call.span }; +} + +/** + * A body whose only `return` is its last statement: the statements before it, and the value it + * returns. Absent when the body returns early anywhere, which is what the sentinel protocol in + * `flattenSeq` exists for. + */ +function trailingReturnOnly(body: readonly CStmt[]): { before: readonly CStmt[]; value: CExpr } | undefined { + const last = body[body.length - 1]; + if (last === undefined || last.kind !== "return" || last.value === undefined) return undefined; + const before = body.slice(0, -1); + return stmtsContainReturn(before) ? undefined : { before, value: last.value }; +} + +/** + * The shape worth having a fast path for: a callee that is one `return `. + * + * Substituting it is what a reader means by inlining — `digitAt(cpf, index)` becomes + * `cpf.charCodeAt(index) - 48` and nothing else changes. Routing it through the general path + * instead would bind each parameter to a `let`, open a synthetic `Option`, assign through it and + * unwrap it, which is four statements and a sentinel where the source had an expression: bigger + * than the call it replaced, in a target that pays for its output by the byte, and slower to read + * in every target. So this case is handled directly and costs only whatever arguments have to be + * bound. + * + * An argument is substituted straight into the body when it is a literal or a local — evaluating + * it twice, or at a different point in the body, is not observable and costs nothing. Anything + * else is bound to one `let` first, in argument order, because it may have an effect (`env` is + * threaded as an ordinary value, so `randomDigit(env)` is an argument that reads the world) and + * because a parameter read twice would otherwise evaluate it twice. + */ +function tryInlineExpression( + call: Extract, + callee: CFunc, + fresh: (base: string) => string, +): { value: CExpr; hoisted: CStmt[] } | undefined { + if (callee.body.length !== 1) return undefined; + const only = callee.body[0]!; + if (only.kind !== "return" || only.value === undefined) return undefined; + + const hoisted: CStmt[] = []; + const substitution = new Map(); + for (let index = 0; index < callee.params.length; index++) { + const param = callee.params[index]!; + const arg = call.args[index]!; + if (arg.kind === "lit" || arg.kind === "local") { + substitution.set(param.name, arg); + continue; + } + const name = fresh(param.name); + hoisted.push({ kind: "let", name, mutable: false, init: arg, type: param.type, span: call.span }); + substitution.set(param.name, { kind: "local", name, type: param.type, span: call.span }); + } + + return { value: substituteExpr(only.value, substitution), hoisted }; +} + +/** `renameStmt`, but mapping a local onto a whole expression rather than onto another name. Only + * ever called with names that are never assigned to, so no assignment target can need rewriting. */ +function substituteStmt(statement: CStmt, map: ReadonlyMap): CStmt { + switch (statement.kind) { + case "let": + return { ...statement, init: substituteExpr(statement.init, map) }; + case "assign": + return { ...statement, value: substituteExpr(statement.value, map) }; + case "setIndex": + return { ...statement, index: substituteExpr(statement.index, map), value: substituteExpr(statement.value, map) }; + case "push": + return { ...statement, value: substituteExpr(statement.value, map) }; + case "if": + return { + ...statement, + test: substituteExpr(statement.test, map), + then: statement.then.map((item) => substituteStmt(item, map)), + otherwise: statement.otherwise.map((item) => substituteStmt(item, map)), + }; + case "switch": + return { + ...statement, + subject: substituteExpr(statement.subject, map), + cases: statement.cases.map((entry) => ({ ...entry, body: entry.body.map((item) => substituteStmt(item, map)) })), + otherwise: statement.otherwise?.map((item) => substituteStmt(item, map)), + }; + case "forRange": + return { + ...statement, + from: substituteExpr(statement.from, map), + to: substituteExpr(statement.to, map), + body: statement.body.map((item) => substituteStmt(item, map)), + }; + case "forEach": + return { + ...statement, + iterable: substituteExpr(statement.iterable, map), + body: statement.body.map((item) => substituteStmt(item, map)), + }; + case "return": + return statement.value === undefined ? statement : { ...statement, value: substituteExpr(statement.value, map) }; + case "fail": + return { ...statement, args: statement.args.map((arg) => substituteExpr(arg, map)) }; + case "break": + case "continue": + return statement; + case "expr": + return { ...statement, expr: substituteExpr(statement.expr, map) }; + default: { + const exhaustive: never = statement; + return exhaustive; + } + } +} + +/** `renameExpr`, but mapping a local onto a whole expression rather than onto another name. */ +function substituteExpr(expr: CExpr, map: ReadonlyMap): CExpr { + switch (expr.kind) { + case "lit": + case "none": + return expr; + case "local": + return map.get(expr.name) ?? expr; + case "some": + return { ...expr, inner: substituteExpr(expr.inner, map) }; + case "record": + return { ...expr, fields: expr.fields.map((field) => ({ ...field, value: substituteExpr(field.value, map) })) }; + case "field": + return { ...expr, target: substituteExpr(expr.target, map) }; + case "list": + return { ...expr, items: expr.items.map((item) => substituteExpr(item, map)) }; + case "call": + case "op": + return { ...expr, args: expr.args.map((arg) => substituteExpr(arg, map)) }; + case "lambda": + // Excluded by `containsLambda` before a callee ever reaches this function. + return expr; + case "cond": + return { + ...expr, + test: substituteExpr(expr.test, map), + then: substituteExpr(expr.then, map), + otherwise: substituteExpr(expr.otherwise, map), + }; + case "and": + case "or": + return { ...expr, left: substituteExpr(expr.left, map), right: substituteExpr(expr.right, map) }; + case "not": + return { ...expr, operand: substituteExpr(expr.operand, map) }; + default: { + const exhaustive: never = expr; + return exhaustive; + } + } +} + +/* ------------------------------------------------------------------ * + * Recomputing `CFunc.calls` after a body changed + * ------------------------------------------------------------------ */ + +function collectCalls(body: readonly CStmt[]): string[] { + return [...new Set(collectCallsWithRepeats(body))]; +} + +/** Every call in `body`, one entry per call site rather than one per callee. */ +function collectCallsWithRepeats(body: readonly CStmt[]): string[] { + const found: string[] = []; + const inExpr = (expr: CExpr): void => { + switch (expr.kind) { + case "some": + inExpr(expr.inner); + return; + case "record": + expr.fields.forEach((field) => inExpr(field.value)); + return; + case "field": + inExpr(expr.target); + return; + case "list": + expr.items.forEach(inExpr); + return; + case "call": + found.push(expr.fn); + expr.args.forEach(inExpr); + return; + case "op": + expr.args.forEach(inExpr); + return; + case "cond": + inExpr(expr.test); + inExpr(expr.then); + inExpr(expr.otherwise); + return; + case "and": + case "or": + inExpr(expr.left); + inExpr(expr.right); + return; + case "not": + inExpr(expr.operand); + return; + case "lambda": + expr.body.forEach(inStmt); + return; + default: + return; + } + }; + const inStmt = (statement: CStmt): void => { + switch (statement.kind) { + case "let": + inExpr(statement.init); + return; + case "assign": + inExpr(statement.value); + return; + case "setIndex": + inExpr(statement.index); + inExpr(statement.value); + return; + case "push": + inExpr(statement.value); + return; + case "if": + inExpr(statement.test); + statement.then.forEach(inStmt); + statement.otherwise.forEach(inStmt); + return; + case "switch": + inExpr(statement.subject); + statement.cases.forEach((entry) => entry.body.forEach(inStmt)); + statement.otherwise?.forEach(inStmt); + return; + case "forRange": + inExpr(statement.from); + inExpr(statement.to); + statement.body.forEach(inStmt); + return; + case "forEach": + inExpr(statement.iterable); + statement.body.forEach(inStmt); + return; + case "return": + if (statement.value !== undefined) inExpr(statement.value); + return; + case "fail": + statement.args.forEach(inExpr); + return; + case "expr": + inExpr(statement.expr); + return; + default: + return; + } + }; + body.forEach(inStmt); + return found; +} diff --git a/engine/src/optimize/optimize.ts b/engine/src/optimize/optimize.ts new file mode 100644 index 000000000..17b5bfb7e --- /dev/null +++ b/engine/src/optimize/optimize.ts @@ -0,0 +1,709 @@ +/** + * Core-to-Core passes: constant folding, dead code elimination and loop raising. + * + * Every pass here is semantics preserving, and `tests/translation.spec.ts` proves it the way the + * architecture requires: the reference interpreter runs the Core before and after each pass on + * generated inputs, and the results must be identical. + */ + +import { tryEvalConst } from "../comptime/eval.ts"; +import type { CExpr, CFunc, CProgram, CStmt } from "../core/ir.ts"; +import { lookupIntrinsic } from "../intrinsics/index.ts"; +import { tLambda } from "../types.ts"; + +export type OptimizeOptions = { + readonly fold?: boolean; + readonly dce?: boolean; + readonly raiseLoops?: boolean; +}; + +export function optimize(program: CProgram, options: OptimizeOptions = {}): CProgram { + const settings = { fold: true, dce: true, raiseLoops: true, ...options }; + let current = program; + // Twice, because the two halves feed each other: folding a read leaves a binding unread, and + // dropping a parameter leaves whatever the call site passed for it unread in *its* caller. A + // second round settles that; a third has never had anything to find. + for (let round = 0; round < 2; round++) { + const functions = new Map(); + for (const [name, fn] of current.functions) { + let body = fn.body; + if (settings.raiseLoops) body = raiseLoops(body); + if (settings.fold) body = body.map(foldStmt); + if (settings.dce) body = removeUnreadBindings(eliminateDeadCode(body)); + functions.set(name, { ...fn, body }); + } + const perFunction = { ...current, functions }; + const next = settings.dce ? removeUnreadParameters(perFunction) : perFunction; + if (round > 0 && next === perFunction) return next; + current = next; + } + return current; +} + +/* ------------------------------------------------------------------ * + * Parameters nothing reads + * ------------------------------------------------------------------ */ + +/** + * Drops a parameter no call needs and the body never reads, rewriting every call site with it. + * + * The same folding that empties a binding empties a parameter: a helper specialized to a call site + * that passes `10` (ADR 0004) has a parameter of type `Int[10..10]`, every read of it becomes that + * constant, and what is left is an argument passed to nobody — which Go and Rust both refuse to + * compile, and which no author would have written either. + * + * Two conditions, both necessary. An entry point keeps every parameter whatever its body does, + * because its signature is the published API rather than an implementation detail. And a call site + * whose argument could be noticed — a call, a draw, anything `isObservationFree` will not vouch + * for — keeps the parameter for every call site, since dropping it would drop that evaluation. + */ +function removeUnreadParameters(program: CProgram): CProgram { + const entryPoints = new Set(program.entryPoints); + const droppable = new Map>(); + for (const [name, fn] of program.functions) { + if (entryPoints.has(name) || fn.params.length === 0) continue; + const read = new Set(); + countReads(fn.body, read); + const unread = fn.params.flatMap((param, index) => (read.has(param.name) ? [] : [index])); + if (unread.length > 0) droppable.set(name, new Set(unread)); + } + if (droppable.size === 0) return program; + + // One veto from any call site removes the index for every call site: the definition has one + // signature, so a parameter is either dropped everywhere or kept everywhere. + for (const fn of program.functions.values()) { + forEachCall(fn.body, (call) => { + const indices = droppable.get(call.fn); + if (indices === undefined) return; + for (const index of [...indices]) { + const arg = call.args[index]; + if (arg === undefined || !isObservationFree(arg)) indices.delete(index); + } + }); + } + for (const [name, indices] of [...droppable]) if (indices.size === 0) droppable.delete(name); + if (droppable.size === 0) return program; + + const keep = + (indices: ReadonlySet) => + (_item: T, index: number): boolean => + !indices.has(index); + const rewriteCall = (expr: CExpr): CExpr => { + if (expr.kind !== "call") return expr; + const indices = droppable.get(expr.fn); + return indices === undefined ? expr : { ...expr, args: expr.args.filter(keep(indices)) }; + }; + + const functions = new Map(); + for (const [name, fn] of program.functions) { + const indices = droppable.get(name); + const params = indices === undefined ? fn.params : fn.params.filter(keep(indices)); + functions.set(name, { + ...fn, + params, + body: fn.body.map((statement) => mapStmtExprs(statement, rewriteCall)), + calls: fn.calls, + }); + } + return { ...program, functions }; +} + +/** Applies `visit` to every call this body holds, children first. */ +function forEachCall(body: readonly CStmt[], visit: (call: Extract) => void): void { + for (const statement of body) { + mapStmtExprs(statement, (expr) => { + if (expr.kind === "call") visit(expr); + return expr; + }); + } +} + +/** Rewrites every expression of a statement bottom-up, recursing into nested statement lists. */ +function mapStmtExprs(statement: CStmt, visit: (expr: CExpr) => CExpr): CStmt { + const at = (expr: CExpr): CExpr => mapExprDeep(expr, visit); + const inner = (body: readonly CStmt[]): CStmt[] => body.map((item) => mapStmtExprs(item, visit)); + switch (statement.kind) { + case "let": + return { ...statement, init: at(statement.init) }; + case "assign": + return { ...statement, value: at(statement.value) }; + case "setIndex": + return { ...statement, index: at(statement.index), value: at(statement.value) }; + case "push": + return { ...statement, value: at(statement.value) }; + case "if": + return { ...statement, test: at(statement.test), then: inner(statement.then), otherwise: inner(statement.otherwise) }; + case "switch": + return { + ...statement, + subject: at(statement.subject), + cases: statement.cases.map((entry) => ({ ...entry, body: inner(entry.body) })), + otherwise: statement.otherwise === undefined ? undefined : inner(statement.otherwise), + }; + case "forRange": + return { ...statement, from: at(statement.from), to: at(statement.to), body: inner(statement.body) }; + case "forEach": + return { ...statement, iterable: at(statement.iterable), body: inner(statement.body) }; + case "return": + return statement.value === undefined ? statement : { ...statement, value: at(statement.value) }; + case "fail": + return { ...statement, args: statement.args.map(at) }; + case "expr": + return { ...statement, expr: at(statement.expr) }; + default: + return statement; + } +} + +function mapExprDeep(expr: CExpr, visit: (expr: CExpr) => CExpr): CExpr { + const at = (inner: CExpr): CExpr => mapExprDeep(inner, visit); + switch (expr.kind) { + case "lit": + case "local": + case "none": + return visit(expr); + case "some": + return visit({ ...expr, inner: at(expr.inner) }); + case "record": + return visit({ ...expr, fields: expr.fields.map((field) => ({ ...field, value: at(field.value) })) }); + case "field": + return visit({ ...expr, target: at(expr.target) }); + case "list": + return visit({ ...expr, items: expr.items.map(at) }); + case "call": + case "op": + return visit({ ...expr, args: expr.args.map(at) }); + case "lambda": + return visit({ ...expr, body: expr.body.map((item) => mapStmtExprs(item, visit)) }); + case "cond": + return visit({ ...expr, test: at(expr.test), then: at(expr.then), otherwise: at(expr.otherwise) }); + case "and": + case "or": + return visit({ ...expr, left: at(expr.left), right: at(expr.right) }); + case "not": + return visit({ ...expr, operand: at(expr.operand) }); + default: { + const exhaustive: never = expr; + return exhaustive; + } + } +} + +/* ------------------------------------------------------------------ * + * Constant folding + * ------------------------------------------------------------------ */ + +export function foldExpr(expr: CExpr): CExpr { + switch (expr.kind) { + case "local": { + // A read the checker proved to be one integer *is* that integer. Nothing here decides + // that; the range on the node is the checker's own conclusion at this read, and this pass + // only stops printing a name for something already known. It matters after + // specialization (ADR 0004), which is what gives a helper called with a constant a + // parameter of type `Int[n..n]`: `randomBelow`'s `4294967296 % bound` is a modulo per + // draw while `bound` is a name, and the constant 4294967290 once it is not. + const type = expr.type; + if (type.kind !== "Int" || type.lo !== type.hi) return expr; + return { kind: "lit", value: type.lo, type, span: expr.span }; + } + case "op": { + const args = expr.args.map(foldExpr); + const definition = lookupIntrinsic(expr.op); + const foldable = + definition !== undefined && + definition.comptime && + expr.op !== "opt.unwrap" && + args.every((arg) => arg.kind === "lit" || arg.kind === "none"); + if (foldable) { + const value = tryEvalConst({ ...expr, args }); + if (value !== undefined && (typeof value !== "object" || Array.isArray(value))) { + return { kind: "lit", value, type: expr.type, span: expr.span }; + } + } + return { ...expr, args }; + } + case "not": { + const operand = foldExpr(expr.operand); + if (operand.kind === "lit" && typeof operand.value === "boolean") { + return { kind: "lit", value: !operand.value, type: expr.type, span: expr.span }; + } + return { ...expr, operand }; + } + case "and": { + const left = foldExpr(expr.left); + const right = foldExpr(expr.right); + if (left.kind === "lit" && left.value === false) return left; + if (left.kind === "lit" && left.value === true) return right; + return { ...expr, left, right }; + } + case "or": { + const left = foldExpr(expr.left); + const right = foldExpr(expr.right); + if (left.kind === "lit" && left.value === true) return left; + if (left.kind === "lit" && left.value === false) return right; + return { ...expr, left, right }; + } + case "cond": { + const test = foldExpr(expr.test); + const then = foldExpr(expr.then); + const otherwise = foldExpr(expr.otherwise); + if (test.kind === "lit") return test.value === true ? then : otherwise; + return { ...expr, test, then, otherwise }; + } + case "call": + return { ...expr, args: expr.args.map(foldExpr) }; + case "record": + return { ...expr, fields: expr.fields.map((field) => ({ ...field, value: foldExpr(field.value) })) }; + case "list": + return { ...expr, items: expr.items.map(foldExpr) }; + case "field": + return { ...expr, target: foldExpr(expr.target) }; + case "some": { + const inner = foldExpr(expr.inner); + // `some(unwrap(x))` is `x`: the checker inserts the unwrap where it proved the value + // present, and re-wrapping it is a round trip every target would otherwise print. + if (inner.kind === "op" && inner.op === "opt.unwrap") return inner.args[0]!; + return { ...expr, inner }; + } + case "lambda": + return { ...expr, body: expr.body.map(foldStmt) }; + default: + return expr; + } +} + +function foldStmt(statement: CStmt): CStmt { + switch (statement.kind) { + case "let": + return { ...statement, init: foldExpr(statement.init) }; + case "assign": + return { ...statement, value: foldExpr(statement.value) }; + case "setIndex": + return { ...statement, index: foldExpr(statement.index), value: foldExpr(statement.value) }; + case "push": + return { ...statement, value: foldExpr(statement.value) }; + case "if": + return { + ...statement, + test: foldExpr(statement.test), + then: statement.then.map(foldStmt), + otherwise: statement.otherwise.map(foldStmt), + }; + case "switch": + return { + ...statement, + subject: foldExpr(statement.subject), + cases: statement.cases.map((entry) => ({ ...entry, body: entry.body.map(foldStmt) })), + otherwise: statement.otherwise?.map(foldStmt), + }; + case "forRange": + return { + ...statement, + from: foldExpr(statement.from), + to: foldExpr(statement.to), + body: statement.body.map(foldStmt), + }; + case "forEach": + return { + ...statement, + iterable: foldExpr(statement.iterable), + body: statement.body.map(foldStmt), + }; + case "return": + return statement.value === undefined ? statement : { ...statement, value: foldExpr(statement.value) }; + case "fail": + return { ...statement, args: statement.args.map(foldExpr) }; + case "expr": + return { ...statement, expr: foldExpr(statement.expr) }; + default: + return statement; + } +} + +/* ------------------------------------------------------------------ * + * Dead code + * ------------------------------------------------------------------ */ + +function eliminateDeadCode(body: readonly CStmt[]): CStmt[] { + const result: CStmt[] = []; + for (const statement of body) { + const rewritten = ((): CStmt => { + switch (statement.kind) { + case "if": { + if (statement.test.kind === "lit" && typeof statement.test.value === "boolean") { + const taken = statement.test.value ? statement.then : statement.otherwise; + return { kind: "if", test: statement.test, then: eliminateDeadCode(taken), otherwise: [], span: statement.span }; + } + return { + ...statement, + then: eliminateDeadCode(statement.then), + otherwise: eliminateDeadCode(statement.otherwise), + }; + } + case "switch": + return { + ...statement, + cases: statement.cases.map((entry) => ({ ...entry, body: eliminateDeadCode(entry.body) })), + otherwise: statement.otherwise === undefined ? undefined : eliminateDeadCode(statement.otherwise), + }; + case "forRange": + case "forEach": + return { ...statement, body: eliminateDeadCode(statement.body) }; + default: + return statement; + } + })(); + result.push(rewritten); + if (rewritten.kind === "return" || rewritten.kind === "fail" || rewritten.kind === "break") break; + } + return result; +} + +/* ------------------------------------------------------------------ * + * Bindings nothing reads + * ------------------------------------------------------------------ */ + +/** + * Drops a `let` whose name is read nowhere and whose initializer cannot be observed. + * + * Folding is what creates these: once every read of a local is replaced by the constant the + * checker proved it to be, the binding itself has no readers left. `easterDayOfMarch`, specialized + * to the years `easterSunday` accepts, proves two of the Meeus algorithm's intermediates constant + * that way, and the declarations they leave behind are not cosmetic — Go and Rust both refuse to + * compile a program with an unused local. + * + * "Cannot be observed" is deliberately conservative: an initializer holding a call, a lambda or + * any intrinsic the registry does not mark comptime is left alone, so a draw from the environment + * or anything that fails is never removed for having an unused result. Repeated to a fixpoint, + * because dropping one binding can leave the one above it unread. + */ +function removeUnreadBindings(body: readonly CStmt[]): CStmt[] { + let current = [...body]; + for (let round = 0; round < 8; round++) { + const read = new Set(); + countReads(current, read); + const next = dropUnread(current, read); + if (next.length === current.length && next.every((statement, index) => statement === current[index])) break; + current = next; + } + return current; +} + +function dropUnread(body: readonly CStmt[], read: ReadonlySet): CStmt[] { + const result: CStmt[] = []; + for (const statement of body) { + switch (statement.kind) { + case "let": + if (!read.has(statement.name) && isObservationFree(statement.init)) continue; + result.push(statement); + break; + case "if": + result.push({ ...statement, then: dropUnread(statement.then, read), otherwise: dropUnread(statement.otherwise, read) }); + break; + case "switch": + result.push({ + ...statement, + cases: statement.cases.map((entry) => ({ ...entry, body: dropUnread(entry.body, read) })), + otherwise: statement.otherwise === undefined ? undefined : dropUnread(statement.otherwise, read), + }); + break; + case "forRange": + case "forEach": + result.push({ ...statement, body: dropUnread(statement.body, read) }); + break; + default: + result.push(statement); + break; + } + } + return result; +} + +/** Every local name `body` reads. A `let`'s own name is not a read of itself; its initializer is. */ +function countReads(body: readonly CStmt[], read: Set): void { + const inExpr = (expr: CExpr): void => { + switch (expr.kind) { + case "local": + read.add(expr.name); + return; + case "some": + inExpr(expr.inner); + return; + case "record": + expr.fields.forEach((field) => inExpr(field.value)); + return; + case "field": + inExpr(expr.target); + return; + case "list": + expr.items.forEach(inExpr); + return; + case "call": + case "op": + expr.args.forEach(inExpr); + return; + case "cond": + inExpr(expr.test); + inExpr(expr.then); + inExpr(expr.otherwise); + return; + case "and": + case "or": + inExpr(expr.left); + inExpr(expr.right); + return; + case "not": + inExpr(expr.operand); + return; + case "lambda": + countReads(expr.body, read); + return; + default: + return; + } + }; + for (const statement of body) { + switch (statement.kind) { + case "let": + inExpr(statement.init); + break; + case "assign": + // The target is written, not read, but a name assigned to still has to keep its + // binding: dropping the `let` would leave the assignment without a declaration. + read.add(statement.name); + inExpr(statement.value); + break; + case "setIndex": + read.add(statement.name); + inExpr(statement.index); + inExpr(statement.value); + break; + case "push": + read.add(statement.name); + inExpr(statement.value); + break; + case "if": + inExpr(statement.test); + countReads(statement.then, read); + countReads(statement.otherwise, read); + break; + case "switch": + inExpr(statement.subject); + statement.cases.forEach((entry) => countReads(entry.body, read)); + if (statement.otherwise !== undefined) countReads(statement.otherwise, read); + break; + case "forRange": + inExpr(statement.from); + inExpr(statement.to); + countReads(statement.body, read); + break; + case "forEach": + inExpr(statement.iterable); + countReads(statement.body, read); + break; + case "return": + if (statement.value !== undefined) inExpr(statement.value); + break; + case "fail": + statement.args.forEach(inExpr); + break; + case "expr": + inExpr(statement.expr); + break; + default: + break; + } + } +} + +/** Whether evaluating this expression can be noticed: a call, a lambda or a non-comptime intrinsic. */ +function isObservationFree(expr: CExpr): boolean { + switch (expr.kind) { + case "lit": + case "local": + case "none": + return true; + case "some": + return isObservationFree(expr.inner); + case "record": + return expr.fields.every((field) => isObservationFree(field.value)); + case "field": + return isObservationFree(expr.target); + case "list": + return expr.items.every(isObservationFree); + case "op": + return (lookupIntrinsic(expr.op)?.comptime ?? false) && expr.args.every(isObservationFree); + case "cond": + return isObservationFree(expr.test) && isObservationFree(expr.then) && isObservationFree(expr.otherwise); + case "and": + case "or": + return isObservationFree(expr.left) && isObservationFree(expr.right); + case "not": + return isObservationFree(expr.operand); + case "call": + case "lambda": + return false; + default: + return false; + } +} + +/* ------------------------------------------------------------------ * + * Loop raising + * ------------------------------------------------------------------ */ + +/** + * Raises the one shape that is unambiguous: an accumulator updated once per element becomes a + * fold. Authors may write either style; the Core never picks a target idiom, and each backend + * decides later whether to print a fold as a loop, a comprehension or a builtin. + */ +function raiseLoops(body: readonly CStmt[]): CStmt[] { + const result: CStmt[] = []; + for (let index = 0; index < body.length; index++) { + const statement = body[index]!; + const next = body[index + 1]; + if ( + statement.kind === "let" && + statement.mutable && + next !== undefined && + next.kind === "forEach" && + next.body.length === 1 && + next.body[0]!.kind === "assign" && + next.body[0]!.name === statement.name && + !mentionsOutsideFold(next.body[0]!.value, statement.name, next.name) && + // A fold is a `const`: sound only when nothing after this loop ever reassigns the same + // local again. Two (or more) loops over the same accumulator, one raisable and one not, + // are an ordinary shape — an author writes one pass that sums and a later pass that + // adjusts — and raising only the first one while leaving the rest as plain `acc = …` + // would hand every backend a `const` its own later statement reassigns. + !isReassignedLater(body.slice(index + 2), statement.name) + ) { + const update = next.body[0]!; + const folded: CStmt = { + kind: "let", + name: statement.name, + mutable: false, + type: statement.type, + init: { + kind: "op", + op: "seq.fold", + args: [ + next.iterable, + statement.init, + { + kind: "lambda", + params: [ + { name: statement.name, type: statement.type }, + { name: next.name, type: next.type }, + ], + body: [{ kind: "return", value: update.value, span: update.span }], + type: tLambda([statement.type, next.type], statement.type), + span: update.span, + }, + ], + type: statement.type, + span: statement.span, + }, + span: statement.span, + }; + result.push(folded); + index++; + continue; + } + switch (statement.kind) { + case "if": + result.push({ ...statement, then: raiseLoops(statement.then), otherwise: raiseLoops(statement.otherwise) }); + break; + case "forRange": + case "forEach": + result.push({ ...statement, body: raiseLoops(statement.body) }); + break; + case "switch": + result.push({ + ...statement, + cases: statement.cases.map((entry) => ({ ...entry, body: raiseLoops(entry.body) })), + otherwise: statement.otherwise === undefined ? undefined : raiseLoops(statement.otherwise), + }); + break; + default: + result.push(statement); + } + } + return result; +} + +/** The update may only mention the accumulator and the element, or the fold would change meaning. */ +function mentionsOutsideFold(expr: CExpr, accumulator: string, element: string): boolean { + let bad = false; + const visit = (node: CExpr): void => { + switch (node.kind) { + case "local": + if (node.name !== accumulator && node.name !== element) bad = true; + return; + case "op": + case "call": + node.args.forEach(visit); + return; + case "record": + node.fields.forEach((field) => visit(field.value)); + return; + case "list": + node.items.forEach(visit); + return; + case "field": + visit(node.target); + return; + case "some": + visit(node.inner); + return; + case "cond": + visit(node.test); + visit(node.then); + visit(node.otherwise); + return; + case "and": + case "or": + visit(node.left); + visit(node.right); + return; + case "not": + visit(node.operand); + return; + case "lambda": + bad = true; + return; + default: + return; + } + }; + visit(expr); + return bad; +} + +/** + * Whether `name` is ever assigned again in `body`, including inside a nested `if`, loop or + * `switch`. A lambda's own body is never checked: the subset rejects a closure that captures a + * mutable local (`docs/semantics.md` section 7), so a lambda can never be the reassignment this + * is looking for. + */ +function isReassignedLater(body: readonly CStmt[], name: string): boolean { + return body.some((statement): boolean => { + switch (statement.kind) { + case "assign": + case "setIndex": + return statement.name === name; + case "if": + return isReassignedLater(statement.then, name) || isReassignedLater(statement.otherwise, name); + case "forRange": + case "forEach": + return isReassignedLater(statement.body, name); + case "switch": + return ( + statement.cases.some((entry) => isReassignedLater(entry.body, name)) || + (statement.otherwise !== undefined && isReassignedLater(statement.otherwise, name)) + ); + default: + return false; + } + }); +} diff --git a/engine/src/project.ts b/engine/src/project.ts new file mode 100644 index 000000000..5c8642195 --- /dev/null +++ b/engine/src/project.ts @@ -0,0 +1,82 @@ +/** + * Project loading: the engine is generic, so everything project-specific lives in a config file + * next to the sources. Nothing in the compiler knows what a project's utilities are about. + */ + +import { readFileSync, readdirSync, statSync } from "node:fs"; +import { join, relative, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; +import { Diagnostics } from "./diagnostics.ts"; +import { parseModule } from "./frontend/lower.ts"; +import type { HModule } from "./hir/ast.ts"; + +export type TargetName = "typescript" | "python" | "go" | "rust"; + +export type ProjectConfig = { + /** Human readable name, used in generated file headers. */ + readonly name: string; + /** Directory holding the source-language modules, relative to the config file. */ + readonly sourceRoot: string; + /** Where generated code goes, relative to the config file. */ + readonly out: string; + readonly targets: readonly TargetName[]; + /** Package or module prefix each target uses for the generated core. */ + readonly packages?: Readonly>>; +}; + +export const DEFAULT_CONFIG: ProjectConfig = { + name: "project", + sourceRoot: "source", + out: "out", + targets: ["typescript", "python", "go", "rust"], +}; + +export function loadConfig(path: string): { config: ProjectConfig; root: string } { + const file = resolve(path); + const raw = JSON.parse(readFileSync(file, "utf8")) as Partial; + return { + config: { ...DEFAULT_CONFIG, ...raw }, + root: resolve(file, ".."), + }; +} + +export function sourceFiles(root: string): string[] { + const files: string[] = []; + const walk = (directory: string): void => { + for (const entry of readdirSync(directory).sort()) { + const full = join(directory, entry); + if (statSync(full).isDirectory()) { + walk(full); + continue; + } + if (entry.endsWith(".ts") && !entry.endsWith(".d.ts")) files.push(full); + } + }; + walk(root); + return files; +} + +/** + * The engine's own source-language standard library. + * + * Portable lowerings call these, so a target is complete as soon as its core constructs lower: + * a native lowering is an optimization, never a requirement. They are compiled like any other + * module and pruned when nothing uses them. + */ +export const STDLIB_ROOT = fileURLToPath(new URL("../stdlib", import.meta.url)); + +/** Parses every module of a project, plus the engine's standard library, into HIR. */ +export function loadModules(sourceRoot: string, diagnostics: Diagnostics): HModule[] { + const modules: HModule[] = []; + for (const file of sourceFiles(STDLIB_ROOT)) { + const source = readFileSync(file, "utf8"); + const name = relative(STDLIB_ROOT, file).replace(/\.ts$/, "").split("\\").join("/"); + modules.push(parseModule(file, `std/${name}`, source, diagnostics)); + } + for (const file of sourceFiles(sourceRoot)) { + const source = readFileSync(file, "utf8"); + const path = relative(sourceRoot, file).replace(/\.ts$/, "").split("\\").join("/"); + modules.push(parseModule(file, path, source, diagnostics)); + } + return modules; +} diff --git a/engine/src/regex.ts b/engine/src/regex.ts new file mode 100644 index 000000000..153818f47 --- /dev/null +++ b/engine/src/regex.ts @@ -0,0 +1,491 @@ +/** + * The regex subset: parsed at compile time, normalized into explicit code point classes, and + * never a run-time value. + * + * Rejected by construction: lookaround, backreferences, the implicit Unicode classes (`\d`, `\w`, + * `\s`, `\b`, which match different sets in JavaScript, Python and Go), inline flags, and any + * pattern that is not anchored to the whole string. What survives normalization means exactly the + * same thing in every target dialect, which is what makes the native lowering safe. + */ + +import type { Span } from "./diagnostics.ts"; +import { MAX_COLLECTION_LENGTH } from "./types.ts"; + +export type CharRange = { readonly lo: number; readonly hi: number }; + +export type RegexNode = + /** A set of code points, always stored as sorted, non-overlapping ranges. */ + | { readonly kind: "class"; readonly ranges: readonly CharRange[]; readonly negated: boolean } + | { readonly kind: "seq"; readonly items: readonly RegexNode[] } + | { readonly kind: "alt"; readonly options: readonly RegexNode[] } + | { + readonly kind: "repeat"; + readonly item: RegexNode; + readonly min: number; + /** `null` means unbounded. */ + readonly max: number | null; + }; + +export type NormalizedRegex = { + /** The original source text, kept for diagnostics and for `LOWERING.md`. */ + readonly source: string; + readonly node: RegexNode; + /** True when every code point the pattern can match is below 0x80. */ + readonly asciiOnly: boolean; + /** True when the pattern only ever matches strings of ASCII digits. */ + readonly digitsOnly: boolean; + /** Proven length range of any string the pattern matches. */ + readonly minLength: number; + readonly maxLength: number; +}; + +export class RegexError extends Error { + readonly span: Span | undefined; + + constructor(message: string, span?: Span) { + super(message); + this.name = "RegexError"; + this.span = span; + } +} + +const MAX_CODE_POINT = 0x10ffff; + +function normalizeRanges(ranges: readonly CharRange[]): CharRange[] { + const sorted = [...ranges].sort((a, b) => a.lo - b.lo || a.hi - b.hi); + const merged: CharRange[] = []; + for (const range of sorted) { + const last = merged[merged.length - 1]; + if (last !== undefined && range.lo <= last.hi + 1) { + merged[merged.length - 1] = { lo: last.lo, hi: Math.max(last.hi, range.hi) }; + } else { + merged.push({ ...range }); + } + } + return merged; +} + +/** + * Expands a negated class into positive ranges, so every class is positive after parsing. + * Exported for the Rust target's chain-scanner classifier (`engine/src/targets/rust/index.ts`), + * which needs the same complement to decide whether two classes can ever overlap; see that file's + * "Regex" section for why. + */ +export function complement(ranges: readonly CharRange[]): CharRange[] { + const result: CharRange[] = []; + let cursor = 0; + for (const range of normalizeRanges(ranges)) { + if (range.lo > cursor) result.push({ lo: cursor, hi: range.lo - 1 }); + cursor = Math.max(cursor, range.hi + 1); + } + if (cursor <= MAX_CODE_POINT) result.push({ lo: cursor, hi: MAX_CODE_POINT }); + return result; +} + +class Parser { + private readonly text: string; + private index = 0; + private readonly span: Span | undefined; + + constructor(text: string, span?: Span) { + this.text = text; + this.span = span; + } + + private fail(message: string): never { + throw new RegexError(`${message} (at offset ${this.index} of /${this.text}/)`, this.span); + } + + private peek(): string | undefined { + return this.text[this.index]; + } + + private eat(char: string): boolean { + if (this.text[this.index] === char) { + this.index++; + return true; + } + return false; + } + + parse(): RegexNode { + if (!this.eat("^")) { + this.fail("a pattern must be anchored with ^ at the start"); + } + const node = this.parseAlt(); + if (!this.eat("$")) { + this.fail("a pattern must be anchored with $ at the end"); + } + if (this.index !== this.text.length) this.fail("trailing characters after $"); + return node; + } + + private parseAlt(): RegexNode { + const options: RegexNode[] = [this.parseSeq()]; + while (this.eat("|")) options.push(this.parseSeq()); + return options.length === 1 ? options[0]! : { kind: "alt", options }; + } + + private parseSeq(): RegexNode { + const items: RegexNode[] = []; + for (;;) { + const next = this.peek(); + if (next === undefined || next === "|" || next === ")" || next === "$") break; + items.push(this.parseRepeat()); + } + return items.length === 1 ? items[0]! : { kind: "seq", items }; + } + + private parseRepeat(): RegexNode { + const atom = this.parseAtom(); + const next = this.peek(); + if (next === "*") { + this.index++; + this.rejectLazy(); + return { kind: "repeat", item: atom, min: 0, max: null }; + } + if (next === "+") { + this.index++; + this.rejectLazy(); + return { kind: "repeat", item: atom, min: 1, max: null }; + } + if (next === "?") { + this.index++; + this.rejectLazy(); + return { kind: "repeat", item: atom, min: 0, max: 1 }; + } + if (next === "{") { + const closing = this.text.indexOf("}", this.index); + if (closing < 0) this.fail("unterminated {n,m}"); + const body = this.text.slice(this.index + 1, closing); + this.index = closing + 1; + this.rejectLazy(); + const parts = body.split(","); + const min = Number.parseInt(parts[0] ?? "", 10); + if (!Number.isInteger(min)) this.fail("{n,m} needs an integer lower bound"); + const max = + parts.length === 1 + ? min + : parts[1] === "" + ? null + : Number.parseInt(parts[1] ?? "", 10); + if (max !== null && !Number.isInteger(max)) this.fail("{n,m} needs an integer upper bound"); + return { kind: "repeat", item: atom, min, max }; + } + return atom; + } + + private rejectLazy(): void { + if (this.peek() === "?") { + this.fail("lazy quantifiers are outside the subset: engines differ on their interaction with anchoring"); + } + } + + private parseAtom(): RegexNode { + if (this.eat("(")) { + if (this.text.startsWith("?", this.index)) { + if (!this.text.startsWith("?:", this.index)) { + this.fail("only non-capturing groups (?:…) are supported; lookaround is outside the subset"); + } + this.index += 2; + } + const node = this.parseAlt(); + if (!this.eat(")")) this.fail("unterminated group"); + return node; + } + if (this.eat("[")) return this.parseClass(); + if (this.eat(".")) { + this.fail("`.` is outside the subset: it means different sets with and without the s flag; write an explicit class"); + } + const char = this.peek(); + if (char === undefined) this.fail("unexpected end of pattern"); + if (char === "\\") { + this.index++; + const point = this.parseEscape(); + return { kind: "class", ranges: [{ lo: point, hi: point }], negated: false }; + } + this.index++; + const point = char.codePointAt(0)!; + if (point > 0xffff) this.index++; + return { kind: "class", ranges: [{ lo: point, hi: point }], negated: false }; + } + + private parseEscape(): number { + const char = this.text[this.index]; + if (char === undefined) this.fail("dangling escape"); + this.index++; + switch (char) { + case "d": + case "D": + case "w": + case "W": + case "s": + case "S": + case "b": + case "B": + this.fail( + `\\${char} is outside the subset: it matches a different set in JavaScript, Python and Go. Write the explicit class instead`, + ); + break; + case "n": + return 0x0a; + case "r": + return 0x0d; + case "t": + return 0x09; + case "u": { + if (this.text[this.index] === "{") { + const closing = this.text.indexOf("}", this.index); + if (closing < 0) this.fail("unterminated \\u{...}"); + const point = Number.parseInt(this.text.slice(this.index + 1, closing), 16); + this.index = closing + 1; + return point; + } + const point = Number.parseInt(this.text.slice(this.index, this.index + 4), 16); + this.index += 4; + return point; + } + default: + return char.codePointAt(0)!; + } + return 0; + } + + private parseClass(): RegexNode { + const negated = this.eat("^"); + const ranges: CharRange[] = []; + for (;;) { + const char = this.peek(); + if (char === undefined) this.fail("unterminated character class"); + if (char === "]") { + this.index++; + break; + } + let lo: number; + if (char === "\\") { + this.index++; + lo = this.parseEscape(); + } else { + this.index++; + lo = char.codePointAt(0)!; + if (lo > 0xffff) this.index++; + } + if (this.peek() === "-" && this.text[this.index + 1] !== "]") { + this.index++; + const next = this.peek(); + if (next === undefined) this.fail("unterminated range"); + let hi: number; + if (next === "\\") { + this.index++; + hi = this.parseEscape(); + } else { + this.index++; + hi = next.codePointAt(0)!; + if (hi > 0xffff) this.index++; + } + ranges.push({ lo, hi }); + } else { + ranges.push({ lo, hi: lo }); + } + } + const positive = negated ? complement(ranges) : normalizeRanges(ranges); + return { kind: "class", ranges: positive, negated: false }; + } +} + +function lengthBounds(node: RegexNode): { min: number; max: number } { + switch (node.kind) { + case "class": + return { min: 1, max: 1 }; + case "seq": { + let min = 0; + let max = 0; + for (const item of node.items) { + const bounds = lengthBounds(item); + min += bounds.min; + max = Math.min(max + bounds.max, MAX_COLLECTION_LENGTH); + } + return { min, max }; + } + case "alt": { + const bounds = node.options.map(lengthBounds); + return { + min: Math.min(...bounds.map((bound) => bound.min)), + max: Math.max(...bounds.map((bound) => bound.max)), + }; + } + case "repeat": { + const inner = lengthBounds(node.item); + return { + min: inner.min * node.min, + max: + node.max === null + ? MAX_COLLECTION_LENGTH + : Math.min(inner.max * node.max, MAX_COLLECTION_LENGTH), + }; + } + default: { + const exhaustive: never = node; + return exhaustive; + } + } +} + +function everyClass(node: RegexNode, predicate: (ranges: readonly CharRange[]) => boolean): boolean { + switch (node.kind) { + case "class": + return predicate(node.ranges); + case "seq": + return node.items.every((item) => everyClass(item, predicate)); + case "alt": + return node.options.every((option) => everyClass(option, predicate)); + case "repeat": + return node.min === 0 && node.max === 0 ? true : everyClass(node.item, predicate); + default: { + const exhaustive: never = node; + return exhaustive; + } + } +} + +export function normalizeRegex(source: string, span?: Span): NormalizedRegex { + const node = new Parser(source, span).parse(); + const bounds = lengthBounds(node); + return { + source, + node, + asciiOnly: everyClass(node, (ranges) => ranges.every((range) => range.hi < 0x80)), + digitsOnly: everyClass(node, (ranges) => + ranges.every((range) => range.lo >= 0x30 && range.hi <= 0x39), + ), + minLength: bounds.min, + maxLength: bounds.max, + }; +} + +/** Reference matcher: a full match over the scalars of the input. */ +export function regexMatches(regex: NormalizedRegex, input: readonly number[]): boolean { + return matchNode(regex.node, input, 0, (next) => next === input.length); +} + +function matchNode( + node: RegexNode, + input: readonly number[], + position: number, + cont: (next: number) => boolean, +): boolean { + switch (node.kind) { + case "class": { + const point = input[position]; + if (point === undefined) return false; + const inSet = node.ranges.some((range) => point >= range.lo && point <= range.hi); + return inSet ? cont(position + 1) : false; + } + case "seq": { + const step = (index: number, at: number): boolean => { + if (index === node.items.length) return cont(at); + return matchNode(node.items[index]!, input, at, (next) => step(index + 1, next)); + }; + return step(0, position); + } + case "alt": + return node.options.some((option) => matchNode(option, input, position, cont)); + case "repeat": { + const limit = node.max ?? Number.MAX_SAFE_INTEGER; + const step = (count: number, at: number): boolean => { + // Greedy: try one more repetition first, then fall back to continuing. + if (count < limit) { + const consumed = matchNode(node.item, input, at, (next) => + next === at ? false : step(count + 1, next), + ); + if (consumed) return true; + } + return count >= node.min ? cont(at) : false; + }; + return step(0, position); + } + default: { + const exhaustive: never = node; + return exhaustive; + } + } +} + +/** + * Prints the normalized pattern in a dialect every target reads the same way: explicit classes, + * no shorthand, and anchoring supplied by the caller (`\A…\z` in Go, `fullmatch` in Python, + * `^…$` in JavaScript, whose `^`/`$` are string anchors without the `m` flag). + */ +/** The dialects the subset prints into. They read the normalized pattern identically. */ +export type RegexDialect = "javascript" | "python" | "go"; + +export function printRegex(node: RegexNode, dialect: RegexDialect = "javascript"): string { + switch (node.kind) { + case "class": { + if (node.ranges.length === 1 && node.ranges[0]!.lo === node.ranges[0]!.hi) { + return escapeLiteral(node.ranges[0]!.lo, dialect); + } + const body = node.ranges + .map((range) => + range.lo === range.hi + ? escapeInClass(range.lo, dialect) + : `${escapeInClass(range.lo, dialect)}-${escapeInClass(range.hi, dialect)}`, + ) + .join(""); + return `[${body}]`; + } + case "seq": + return node.items.map((item) => printRegex(item, dialect)).join(""); + case "alt": + return `(?:${node.options.map((option) => printRegex(option, dialect)).join("|")})`; + case "repeat": { + const inner = needsGroup(node.item) + ? `(?:${printRegex(node.item, dialect)})` + : printRegex(node.item, dialect); + if (node.min === 0 && node.max === null) return `${inner}*`; + if (node.min === 1 && node.max === null) return `${inner}+`; + if (node.min === 0 && node.max === 1) return `${inner}?`; + if (node.max === null) return `${inner}{${node.min},}`; + if (node.min === node.max) return `${inner}{${node.min}}`; + return `${inner}{${node.min},${node.max}}`; + } + default: { + const exhaustive: never = node; + return exhaustive; + } + } +} + +function needsGroup(node: RegexNode): boolean { + return node.kind === "seq" || node.kind === "alt"; +} + +const SPECIAL = new Set([..."\\^$.|?*+()[]{}"].map((char) => char.codePointAt(0)!)); + +function escapeLiteral(point: number, dialect: RegexDialect): string { + if (SPECIAL.has(point)) return `\\${String.fromCodePoint(point)}`; + return printable(point, dialect); +} + +function escapeInClass(point: number, dialect: RegexDialect): string { + if (point === 0x5d || point === 0x5c || point === 0x2d || point === 0x5e) { + return `\\${String.fromCodePoint(point)}`; + } + return printable(point, dialect); +} + +/** + * A scalar outside printable ASCII is escaped in the dialect's own syntax: RE2 spells a code + * point `\x{…}` and rejects `\uXXXX`, while JavaScript and Python read `\uXXXX`. + */ +function printable(point: number, dialect: RegexDialect): string { + if (point >= 0x20 && point <= 0x7e) return String.fromCodePoint(point); + if (dialect === "go") return `\\x{${point.toString(16)}}`; + if (point < 0x20 || point === 0x7f) return `\\x${point.toString(16).padStart(2, "0")}`; + if (point > 0xffff) { + return dialect === "python" + ? `\\U${point.toString(16).padStart(8, "0")}` + : `\\u{${point.toString(16)}}`; + } + return `\\u${point.toString(16).padStart(4, "0")}`; +} diff --git a/engine/src/stdlib.ts b/engine/src/stdlib.ts new file mode 100644 index 000000000..ca4ff6364 --- /dev/null +++ b/engine/src/stdlib.ts @@ -0,0 +1,19 @@ +/** + * The standard library's public surface. + * + * These functions are checked against their declared signature and are always available to a + * portable lowering, whether or not any project source names them. Everything else under + * `stdlib/` is an internal helper, specialized per call site like any other library code. + * + * The list is target-agnostic on purpose: nothing before the backends may branch on a target, so + * the compiler keeps the whole surface rather than asking which lowerings a target selected. + */ +export const STDLIB_SURFACE: readonly string[] = [ + "std/strings::compareScalars", + "std/strings::asciiUpperAll", + "std/strings::asciiLowerAll", + "std/date::ymdToDays", + "std/date::yearFromDays", + "std/date::monthFromDays", + "std/date::dayFromDays", +]; diff --git a/engine/src/targets/go/index.ts b/engine/src/targets/go/index.ts new file mode 100644 index 000000000..97a457512 --- /dev/null +++ b/engine/src/targets/go/index.ts @@ -0,0 +1,1400 @@ +/** + * The Go backend. + * + * Go 1.21, standard library only, ordinary loops and explicit error returns. A `Fail` effect + * becomes a second return value, so a caller reads exactly like hand written Go; the hoisting + * pass in the lowerer is what turns a nested fallible call into the familiar + * `value, err := f(); if err != nil { return zero, err }`. + * + * `int` is assumed to be 64 bits, which is true on every platform Go supports today except + * 32 bit ones; the assumption is recorded in `docs/targets/go.md`. + */ + +import type { Backend, DriverEntry, SupportNeeds } from "../../backend/generate.ts"; +import { ENGINE_VERSION } from "../../backend/generate.ts"; +import type { TargetSpec } from "../../backend/lower.ts"; +import type { Candidate } from "../../backend/select.ts"; +import { LoweringTable, argIsAscii } from "../../backend/select.ts"; +import { asciiString } from "../../backend/tast.ts"; +import type { TExpr, TFunc, TModule, TRecord, TStmt } from "../../backend/tast.ts"; +import type { CProgram } from "../../core/ir.ts"; +import { printRegex } from "../../regex.ts"; +import { TRIM_CODE_POINTS } from "../../intrinsics/index.ts"; +import type { SemType } from "../../types.ts"; +import type { Value } from "../../values.ts"; + +export const GO_CONFIG = { + baseline: "Go 1.21", + fileExtension: ".go", + packageName: "core", + dependencies: [] as string[], + formatter: "gofmt", + linters: ["go vet", "staticcheck"], +}; + +export function goType(type: SemType): string { + switch (type.kind) { + case "Bool": + return "bool"; + case "Int": + case "Decimal": + case "CivilDate": + case "Instant": + case "Duration": + return "int"; + case "Float": + return "float64"; + case "String": + return "string"; + case "List": + return `[]${goType(type.elem)}`; + case "Option": + return `*${goType(type.inner)}`; + case "Record": + return type.name; + case "Enum": + return "string"; + case "Union": + return type.name; + case "Lambda": + return `func(${type.params.map(goType).join(", ")}) ${goType(type.ret)}`; + case "Void": + return ""; + case "Never": + return "any"; + default: { + const exhaustive: never = type; + return exhaustive; + } + } +} + +export function goZero(type: SemType): string { + switch (type.kind) { + case "Bool": + return "false"; + case "Int": + case "Decimal": + case "CivilDate": + case "Instant": + case "Duration": + return "0"; + case "Float": + return "0"; + case "String": + case "Enum": + return '""'; + case "List": + case "Option": + return "nil"; + case "Record": + return `${type.name}{}`; + default: + return "nil"; + } +} + +const raw = (text: string): TExpr => ({ kind: "raw", text }); + +function binary(op: string): Candidate["emit"] { + return (args) => ({ kind: "binary", op, left: args[0]!, right: args[1]! }); +} + +const cheap = { alloc: "none", time: "constant" } as const; +const linear = { alloc: "none", time: "linear" } as const; +const allocating = { alloc: "one", time: "linear" } as const; +/** A pass that materializes the scalars of a string: correct everywhere, and the slowest option. */ +const scalarPass = { alloc: "many", time: "linear" } as const; + +export const GO_CANDIDATES: readonly Candidate[] = [ + ...["add:+", "sub:-", "mul:*", "div:/", "mod:%"].map((entry) => { + const [op, symbol] = entry.split(":") as [string, string]; + return { + op: `int.${op}`, + impl: "native" as const, + cost: cheap, + because: op === "div" || op === "mod" ? "Go truncates, which is the Core's rule" : undefined, + emit: binary(symbol), + }; + }), + ...["add:+", "sub:-", "mul:*", "div:/"].map((entry) => { + const [op, symbol] = entry.split(":") as [string, string]; + return { op: `float.${op}`, impl: "native" as const, cost: cheap, emit: binary(symbol) }; + }), + { op: "int.neg", impl: "native", cost: cheap, emit: (args) => raw(`-${print(args[0]!)}`) }, + { op: "float.neg", impl: "native", cost: cheap, emit: (args) => raw(`-${print(args[0]!)}`) }, + { + op: "int.abs", + impl: "native", + cost: cheap, + emit: (args) => raw(`func() int { if ${print(args[0]!)} < 0 { return -${print(args[0]!)} }; return ${print(args[0]!)} }()`), + }, + { op: "int.min", impl: "native", cost: cheap, emit: (args) => raw(`min(${print(args[0]!)}, ${print(args[1]!)})`) }, + { op: "int.max", impl: "native", cost: cheap, emit: (args) => raw(`max(${print(args[0]!)}, ${print(args[1]!)})`) }, + ...["lt:<", "le:<=", "gt:>", "ge:>="].flatMap((entry) => { + const [op, symbol] = entry.split(":") as [string, string]; + return [ + { op: `int.${op}`, impl: "native" as const, cost: cheap, emit: binary(symbol) }, + { op: `float.${op}`, impl: "native" as const, cost: cheap, emit: binary(symbol) }, + ]; + }), + { op: "float.fromInt", impl: "native", cost: cheap, emit: (args) => raw(`float64(${print(args[0]!)})`) }, + { op: "core.eq", impl: "native", cost: cheap, emit: binary("==") }, + + { + op: "opt.isNone", + impl: "native", + cost: cheap, + // A binary node rather than a fragment, so negating it prints `!= nil` rather than `!(… == nil)`. + emit: (args) => ({ kind: "binary", op: "==", left: args[0]!, right: raw("nil") }), + }, + { + op: "opt.unwrap", + impl: "native", + cost: cheap, + // Parenthesized: `*x.Field` would dereference the field, not the option. + emit: (args) => raw(`(*${print(args[0]!)})`), + }, + { op: "opt.some", impl: "native", cost: cheap, emit: (args) => raw(`ptr(${print(args[0]!)})`) }, + { + op: "opt.orElse", + impl: "native", + cost: cheap, + emit: (args, types) => + raw(`orElse(${print(args[0]!)}, ${print(args[1]!)})`) as TExpr & { types?: typeof types }, + }, + + { + op: "str.len", + impl: "native", + requires: argIsAscii(0), + because: "`len` counts bytes, which equals the scalar count only for ASCII", + cost: cheap, + emit: (args) => raw(`len(${print(args[0]!)})`), + }, + { + op: "str.len", + impl: "native", + because: "converting to []rune counts scalars, at the cost of one allocation", + cost: allocating, + emit: (args) => raw(`len([]rune(${print(args[0]!)}))`), + }, + { op: "str.concat", impl: "native", cost: allocating, emit: binary("+") }, + { + op: "str.codeAt", + impl: "native", + requires: argIsAscii(0), + because: "indexing a string yields a byte", + cost: cheap, + emit: (args) => raw(`int(${print(args[0]!)}[${print(args[1]!)}])`), + }, + { + op: "str.charAt", + impl: "native", + requires: argIsAscii(0), + cost: allocating, + emit: (args) => raw(`string(${print(args[0]!)}[${print(args[1]!)}])`), + }, + { + op: "str.codeAtOpt", + impl: "library", + requires: argIsAscii(0), + cost: cheap, + emit: (args) => raw(`codeAt(${print(args[0]!)}, ${print(args[1]!)})`), + }, + { + op: "str.charAtOpt", + impl: "library", + requires: argIsAscii(0), + cost: allocating, + emit: (args) => raw(`charAt(${print(args[0]!)}, ${print(args[1]!)})`), + }, + { + op: "str.slice", + impl: "native", + requires: argIsAscii(0), + because: "slicing cuts at byte boundaries", + cost: cheap, + emit: (args) => raw(`${print(args[0]!)}[${print(args[1]!)}:${print(args[2]!)}]`), + }, + { + op: "str.indexOf", + impl: "native", + requires: argIsAscii(0), + because: "strings.Index returns a byte offset", + cost: linear, + deps: ["strings"], + emit: (args, _types, ctx) => { + ctx.require("strings"); + return raw(`strings.Index(${print(args[0]!)}, ${print(args[1]!)})`); + }, + }, + { + op: "str.contains", + impl: "native", + cost: linear, + deps: ["strings"], + emit: (args, _types, ctx) => { + ctx.require("strings"); + return raw(`strings.Contains(${print(args[0]!)}, ${print(args[1]!)})`); + }, + }, + { + op: "str.startsWith", + impl: "native", + cost: linear, + deps: ["strings"], + emit: (args, _types, ctx) => { + ctx.require("strings"); + return raw(`strings.HasPrefix(${print(args[0]!)}, ${print(args[1]!)})`); + }, + }, + { + op: "str.endsWith", + impl: "native", + cost: linear, + deps: ["strings"], + emit: (args, _types, ctx) => { + ctx.require("strings"); + return raw(`strings.HasSuffix(${print(args[0]!)}, ${print(args[1]!)})`); + }, + }, + { + op: "str.repeat", + impl: "native", + cost: allocating, + deps: ["strings"], + emit: (args, _types, ctx) => { + ctx.require("strings"); + return raw(`strings.Repeat(${print(args[0]!)}, ${print(args[1]!)})`); + }, + }, + { + op: "str.padStart", + impl: "library", + cost: allocating, + deps: ["_support"], + emit: (args) => raw(`padStart(${print(args[0]!)}, ${print(args[1]!)}, ${print(args[2]!)})`), + }, + { + op: "str.trim", + impl: "native", + because: "strings.Trim takes the cut set explicitly, so the 25 code points are exact", + cost: cheap, + deps: ["strings"], + emit: (args, _types, ctx) => { + ctx.require("strings"); + return raw( + `strings.Trim(${print(args[0]!)}, ${asciiString(TRIM_CODE_POINTS.map((point) => String.fromCodePoint(point)).join(""))})`, + ); + }, + }, + { + op: "str.asciiUpper", + impl: "native", + requires: argIsAscii(0), + because: "strings.ToUpper is only ASCII-equivalent on ASCII input", + cost: allocating, + deps: ["strings"], + emit: (args, _types, ctx) => { + ctx.require("strings"); + return raw(`strings.ToUpper(${print(args[0]!)})`); + }, + }, + { + op: "str.asciiLower", + impl: "native", + requires: argIsAscii(0), + cost: allocating, + deps: ["strings"], + emit: (args, _types, ctx) => { + ctx.require("strings"); + return raw(`strings.ToLower(${print(args[0]!)})`); + }, + }, + { + op: "str.asciiUpper", + impl: "native", + because: "strings.Map is one pass, and mapping only a-z is the Core's rule for any input", + cost: allocating, + deps: ["strings"], + emit: (args, _types, ctx) => { + ctx.require("strings"); + return raw( + `strings.Map(func(scalar rune) rune { if scalar >= 'a' && scalar <= 'z' { return scalar - 32 }; return scalar }, ${print(args[0]!)})`, + ); + }, + }, + { + op: "str.asciiLower", + impl: "native", + because: "strings.Map is one pass, and mapping only A-Z is the Core's rule for any input", + cost: allocating, + deps: ["strings"], + emit: (args, _types, ctx) => { + ctx.require("strings"); + return raw( + `strings.Map(func(scalar rune) rune { if scalar >= 'A' && scalar <= 'Z' { return scalar + 32 }; return scalar }, ${print(args[0]!)})`, + ); + }, + }, + { + op: "str.compare", + impl: "native", + because: "Go compares UTF-8 bytes, which is code point order", + cost: linear, + deps: ["strings"], + emit: (args, _types, ctx) => { + ctx.require("strings"); + return raw(`strings.Compare(${print(args[0]!)}, ${print(args[1]!)})`); + }, + }, + { + op: "str.codePoints", + impl: "native", + requires: argIsAscii(0), + because: "an ASCII byte is already its own code point, so `[]byte(value)` needs no UTF-8 decode at all, unlike `codePoints`' `range` over the string below", + cost: allocating, + emit: (args) => + raw( + `func() []int { __bs := []byte(${print(args[0]!)}); __pts := make([]int, len(__bs)); for __i, __b := range __bs { __pts[__i] = int(__b) }; return __pts }()`, + ), + }, + { op: "str.codePoints", impl: "library", cost: allocating, emit: (args) => raw(`codePoints(${print(args[0]!)})`) }, + { + op: "str.fromCodePoints", + impl: "native", + requires: (args) => { + const elem = args[0]; + return elem !== undefined && elem.kind === "List" && elem.elem.kind === "Int" && elem.elem.lo >= 0 && elem.elem.hi <= 127; + }, + because: + "every code point this project ever builds this way is proven ASCII (`group_thousands`' `out`, " + + "`engine/docs/progress.md` §8), so its byte value is its whole UTF-8 encoding -- one []byte " + + "built directly and converted once, instead of `fromCodePoints`' []rune round trip below, " + + "which lets Go's own UTF-8 encoder re-derive what a byte already was", + cost: allocating, + emit: (args) => + raw( + `string(func() []byte { __pts := ${print(args[0]!)}; __bs := make([]byte, len(__pts)); for __i, __p := range __pts { __bs[__i] = byte(__p) }; return __bs }())`, + ), + }, + { op: "str.fromCodePoints", impl: "library", cost: allocating, emit: (args) => raw(`fromCodePoints(${print(args[0]!)})`) }, + { op: "str.asAscii", impl: "library", cost: linear, emit: (args) => raw(`asAscii(${print(args[0]!)})`) }, + { op: "str.asDigits", impl: "library", cost: linear, emit: (args) => raw(`asDigits(${print(args[0]!)})`) }, + { + op: "str.split", + impl: "native", + cost: allocating, + deps: ["strings"], + emit: (args, _types, ctx) => { + ctx.require("strings"); + return raw(`strings.Split(${print(args[0]!)}, ${print(args[1]!)})`); + }, + }, + { + op: "str.join", + impl: "native", + cost: allocating, + deps: ["strings"], + emit: (args, _types, ctx) => { + ctx.require("strings"); + return raw(`strings.Join(${print(args[0]!)}, ${print(args[1]!)})`); + }, + }, + { + op: "str.fromInt", + impl: "native", + cost: allocating, + deps: ["strconv"], + emit: (args, _types, ctx) => { + ctx.require("strconv"); + return raw(`strconv.Itoa(${print(args[0]!)})`); + }, + }, + { op: "str.parseInt", impl: "library", cost: linear, emit: (args) => raw(`parseDigits(${print(args[0]!)})`) }, + + { + op: "seq.at", + impl: "library", + cost: cheap, + emit: (args) => raw(`at(${print(args[0]!)}, ${print(args[1]!)})`), + }, + { + op: "str.asciiUpper", + impl: "portable", + // Building a scalar list costs far more than the host's own pass, which is why the cost + // class has to say so: selection ranks by cost before it ranks by implementation kind. + cost: scalarPass, + sourceFn: "std/strings::asciiUpperAll", + emit: (args, _types, ctx) => raw(`${ctx.nameOf("std/strings::asciiUpperAll")}(${print(args[0]!)})`), + }, + { + op: "str.asciiLower", + impl: "portable", + cost: scalarPass, + sourceFn: "std/strings::asciiLowerAll", + emit: (args, _types, ctx) => raw(`${ctx.nameOf("std/strings::asciiLowerAll")}(${print(args[0]!)})`), + }, + { op: "seq.len", impl: "native", cost: cheap, emit: (args) => raw(`len(${print(args[0]!)})`) }, + { op: "seq.get", impl: "native", cost: cheap, emit: (args) => raw(`${print(args[0]!)}[${print(args[1]!)}]`) }, + { op: "seq.push", impl: "native", cost: cheap, emit: (args) => raw(`${print(args[0]!)} = append(${print(args[0]!)}, ${print(args[1]!)})`) }, + { op: "seq.sum", impl: "library", cost: linear, emit: (args) => raw(`sumInts(${print(args[0]!)})`) }, + { op: "seq.contains", impl: "library", cost: linear, emit: (args) => raw(`contains(${print(args[0]!)}, ${print(args[1]!)})`) }, + { op: "seq.indexOf", impl: "library", cost: linear, emit: (args) => raw(`indexOf(${print(args[0]!)}, ${print(args[1]!)})`) }, + { op: "seq.concat", impl: "native", cost: allocating, emit: (args) => raw(`append(append([]${"T"}{}, ${print(args[0]!)}...), ${print(args[1]!)}...)`) }, + { op: "seq.slice", impl: "native", cost: cheap, emit: (args) => raw(`${print(args[0]!)}[${print(args[1]!)}:${print(args[2]!)}]`) }, + { op: "seq.reverse", impl: "library", cost: allocating, emit: (args) => raw(`reversed(${print(args[0]!)})`) }, + { + op: "seq.sortStable", + impl: "native", + because: "slices.SortStableFunc is stable, unlike sort.Slice", + cost: { alloc: "one", time: "nlogn" }, + deps: ["slices"], + emit: (args, _types, ctx) => { + ctx.require("slices"); + return raw(`sortedStable(${print(args[0]!)}, ${print(args[1]!)})`); + }, + }, + { + op: "seq.sortStableBy", + impl: "native", + cost: { alloc: "one", time: "nlogn" }, + deps: ["slices"], + emit: (args, types) => { + const keyType = types[1]; + const compare = + keyType !== undefined && keyType.kind === "Lambda" && keyType.ret.kind === "String" + ? `func(a, b string) int { if a < b { return -1 }; if a > b { return 1 }; return 0 }` + : `func(a, b int) int { return a - b }`; + return raw(`sortedStableBy(${print(args[0]!)}, ${print(args[1]!)}, ${compare})`); + }, + }, + { op: "seq.map", impl: "library", cost: allocating, emit: (args) => raw(`mapped(${print(args[0]!)}, ${print(args[1]!)})`) }, + { op: "seq.filter", impl: "library", cost: allocating, emit: (args) => raw(`filtered(${print(args[0]!)}, ${print(args[1]!)})`) }, + { op: "seq.any", impl: "library", cost: linear, emit: (args) => raw(`anyOf(${print(args[0]!)}, ${print(args[1]!)})`) }, + { op: "seq.all", impl: "library", cost: linear, emit: (args) => raw(`allOf(${print(args[0]!)}, ${print(args[1]!)})`) }, + { op: "seq.find", impl: "library", cost: linear, emit: (args) => raw(`found(${print(args[0]!)}, ${print(args[1]!)})`) }, + + { op: "dec.fromScaled", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { op: "dec.fromInt", impl: "native", cost: cheap, emit: (args, types) => raw(`${print(args[0]!)} * ${10 ** scaleOf(types[1])}`) }, + { op: "dec.add", impl: "native", cost: cheap, emit: binary("+") }, + { op: "dec.sub", impl: "native", cost: cheap, emit: binary("-") }, + { op: "dec.mul", impl: "native", cost: cheap, emit: binary("*") }, + { op: "dec.compare", impl: "library", cost: cheap, emit: (args) => raw(`compareInts(${print(args[0]!)}, ${print(args[1]!)})`) }, + { op: "dec.isNegative", impl: "native", cost: cheap, emit: (args) => raw(`${print(args[0]!)} < 0`) }, + { op: "dec.abs", impl: "library", cost: cheap, emit: (args) => raw(`absInt(${print(args[0]!)})`) }, + { op: "dec.unscaled", impl: "native", cost: cheap, emit: (args) => args[0]! }, + + { + op: "date.clampEpochDays", + impl: "native", + cost: cheap, + emit: (args) => raw(`min(max(${print(args[0]!)}, -719162), 2932896)`), + }, + { op: "date.toEpochDays", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { op: "date.fromEpochDays", impl: "library", cost: cheap, emit: (args) => raw(`dateFromEpochDays(${print(args[0]!)})`) }, + { + op: "date.fromYmd", + impl: "portable", + cost: linear, + sourceFn: "std/date::ymdToDays", + emit: (args, _types, ctx) => raw(`${ctx.nameOf("std/date::ymdToDays")}(${args.map(print).join(", ")})`), + }, + { op: "date.year", impl: "portable", cost: cheap, sourceFn: "std/date::yearFromDays", emit: (args, _types, ctx) => raw(`${ctx.nameOf("std/date::yearFromDays")}(${print(args[0]!)})`) }, + { op: "date.month", impl: "portable", cost: cheap, sourceFn: "std/date::monthFromDays", emit: (args, _types, ctx) => raw(`${ctx.nameOf("std/date::monthFromDays")}(${print(args[0]!)})`) }, + { op: "date.day", impl: "portable", cost: cheap, sourceFn: "std/date::dayFromDays", emit: (args, _types, ctx) => raw(`${ctx.nameOf("std/date::dayFromDays")}(${print(args[0]!)})`) }, + { op: "date.addDays", impl: "library", cost: cheap, emit: (args) => raw(`dateAddDays(${print(args[0]!)}, ${print(args[1]!)})`) }, + { op: "date.diffDays", impl: "native", cost: cheap, emit: binary("-") }, + { op: "date.compare", impl: "library", cost: cheap, emit: (args) => raw(`compareInts(${print(args[0]!)}, ${print(args[1]!)})`) }, + { op: "date.dayOfWeek", impl: "native", cost: cheap, emit: (args) => raw(`(((${print(args[0]!)}+3)%7+7)%7 + 1)`) }, + { + op: "date.isLeapYear", + impl: "native", + cost: cheap, + emit: (args) => raw(`((${print(args[0]!)}%4 == 0 && ${print(args[0]!)}%100 != 0) || ${print(args[0]!)}%400 == 0)`), + }, + + { + // A byte-wise pass, not `strings.Map`, when every retained range is ASCII (every class this + // project uses is: digits, letters). `strings.Map` decodes the whole input as UTF-8 runes + // before the callback ever runs; a byte never needs decoding to be range-tested, and a + // multi-byte scalar's bytes are all >= 0x80, so every one of them fails an ASCII range test + // on its own and is dropped exactly as decoding and testing the scalar would drop it — a + // non-ASCII input keeps working, just without ever paying to decode it + // (`engine/docs/progress.md` §8, same reasoning as the Rust target's `re.retain`). + op: "re.retain", + impl: "native", + because: "a byte-wise scan when every retained range is ASCII", + cost: allocating, + emit: (args, _types, ctx) => { + const ranges = ctx.regex === undefined || ctx.regex.node.kind !== "class" ? [] : ctx.regex.node.ranges; + const asciiTest = (name: string): string => + ranges.length === 0 + ? "false" + : ranges.map((range) => (range.lo === range.hi ? `${name} == ${range.lo}` : `(${name} >= ${range.lo} && ${name} <= ${range.hi})`)).join(" || "); + if (ranges.every((range) => range.hi <= 127)) { + return raw( + `func() string { __value := ${print(args[0]!)}; __out := make([]byte, 0, len(__value)); for __i := 0; __i < len(__value); __i++ { __b := __value[__i]; __c := int(__b); if ${asciiTest("__c")} { __out = append(__out, __b) } }; return string(__out) }()`, + ); + } + ctx.require("strings"); + return raw( + `strings.Map(func(scalar rune) rune { if ${asciiTest("scalar")} { return scalar }; return -1 }, ${print(args[0]!)})`, + ); + }, + }, + { + op: "re.test", + impl: "native", + because: "the normalized pattern is inside the compatibility subset; \\A…\\z anchors the whole string", + cost: linear, + deps: ["regexp"], + emit: (args, _types, ctx) => { + ctx.require("regexp"); + const pattern = ctx.regex === undefined ? "" : printRegex(ctx.regex.node, "go"); + return raw(`regexp.MustCompile(${asciiString(`\\A${pattern}\\z`)}).MatchString(${print(args[0]!)})`); + }, + }, + + { + op: "http.request", + impl: "native", + cost: { alloc: "many", time: "linear" }, + emit: (args, _types, ctx) => raw(`${print(ctx.env())}.Request(${print(args[0]!)})`), + }, + { op: "clock.now", impl: "native", cost: cheap, emit: (_args, _types, ctx) => raw(`${print(ctx.env())}.Now()`) }, + { op: "clock.sleep", impl: "native", cost: cheap, emit: (args, _types, ctx) => raw(`${print(ctx.env())}.Sleep(${print(args[0]!)})`) }, + { op: "clock.millis", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { op: "clock.durationMillis", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { op: "clock.elapsed", impl: "native", cost: cheap, emit: (args) => raw(`max(0, ${print(args[1]!)}-${print(args[0]!)})`) }, + { op: "random.nextU32", impl: "native", cost: cheap, emit: (_args, _types, ctx) => raw(`${print(ctx.env())}.NextU32()`) }, + { + op: "task.race", + impl: "library", + because: "goroutines with a context and a channel are the idiomatic form", + cost: { alloc: "many", time: "linear" }, + emit: (args) => raw(`raceFirstSome(${print(args[0]!)})`), + }, +]; + +function scaleOf(type: SemType | undefined): number { + return type !== undefined && type.kind === "Int" ? Number(type.lo) : 0; +} + +export const GO_SPEC: TargetSpec = { + name: "go", + table: new LoweringTable(GO_CANDIDATES), + naming: { + func: (name, exported) => (exported ? pascal(name) : camel(name)), + value: (name) => camel(name), + field: (name) => pascal(name), + type: (name) => pascal(name), + module: (path) => `${path.split("/").join("_")}.go`, + }, + loopCombinators: new Set(["seq.fold", "seq.map", "seq.filter"]), + statementTernary: true, + errorsAsValues: true, + asyncColouring: false, + envType: { kind: "Record", name: "Capabilities" }, +}; + +function camel(name: string): string { + const cleaned = name.replace(/[-_](.)/g, (_match, char: string) => char.toUpperCase()); + return cleaned.charAt(0).toLowerCase() + cleaned.slice(1); +} + +function pascal(name: string): string { + const cleaned = name.replace(/[-_](.)/g, (_match, char: string) => char.toUpperCase()); + return cleaned.charAt(0).toUpperCase() + cleaned.slice(1); +} + +/* ------------------------------------------------------------------ * + * Printer + * ------------------------------------------------------------------ */ + +export function print(expr: TExpr): string { + switch (expr.kind) { + case "lit": + return literal(expr.value, expr.type); + case "name": + return expr.name; + case "raw": + return expr.text; + case "call": + return `${print(expr.callee)}(${expr.args.map(print).join(", ")})`; + case "method": + return `${print(expr.target)}.${pascal(expr.name)}(${expr.args.map(print).join(", ")})`; + case "member": + return `${print(expr.target)}.${expr.name}`; + case "index": + return `${print(expr.target)}[${print(expr.index)}]`; + case "binary": + return `(${print(expr.left)} ${goOperator(expr.op)} ${print(expr.right)})`; + case "unary": { + // `!(a == b)` reads better as `a != b`, and gofmt cannot do that for us. + if (expr.op === "!" && expr.operand.kind === "binary" && expr.operand.op === "==") { + return `(${print(expr.operand.left)} != ${print(expr.operand.right)})`; + } + // A capability table fragment is opaque, so negating it needs parentheses. + const operand = print(expr.operand); + const atomic = expr.operand.kind === "name" || expr.operand.kind === "call" || expr.operand.kind === "method"; + return atomic ? `${expr.op}${operand}` : `${expr.op}(${operand})`; + } + case "ternary": + return `func() ${"any"} { if ${print(expr.test)} { return ${print(expr.then)} }; return ${print(expr.otherwise)} }()`; + case "list": + return `[]${goType(expr.type.kind === "List" ? expr.type.elem : expr.type)}{${expr.items.map(print).join(", ")}}`; + case "record": + return `${expr.typeName}{${expr.fields.map((field) => `${field.name}: ${print(field.value)}`).join(", ")}}`; + case "lambda": { + const params = expr.params.map((param) => `${param.name} ${goType(param.type)}`).join(", "); + return `func(${params}) ${goType(expr.ret)} {\n${printBody(expr.body, 1)}\n}`; + } + case "none": + return "nil"; + case "zero": + return goZero(expr.type); + case "some": + return `ptr(${print(expr.inner)})`; + default: { + const exhaustive: never = expr; + return exhaustive; + } + } +} + +function goOperator(op: string): string { + switch (op) { + case "===": + return "=="; + case "!==": + return "!="; + case "??": + return "??"; + default: + return op; + } +} + +function literal(value: Value, type?: SemType): string { + if (typeof value === "bigint") return value.toString(); + if (typeof value === "string") return asciiString(value); + if (typeof value === "boolean") return String(value); + if (typeof value === "number") return String(value); + if (Array.isArray(value)) { + const elem = type !== undefined && type.kind === "List" ? type.elem : undefined; + const rendered = elem === undefined ? "int" : goType(elem); + return `[]${rendered}{${value.map((item) => literal(item, elem)).join(", ")}}`; + } + return "nil"; +} + +function printBody(body: readonly TStmt[], depth: number): string { + return body.map((statement) => printStmt(statement, depth)).join("\n"); +} + +function printStmt(statement: TStmt, depth: number): string { + const pad = "\t".repeat(depth); + switch (statement.kind) { + case "let": + return `${pad}${statement.name} := ${print(statement.init)}`; + case "multiLet": + return `${pad}${statement.names.join(", ")} := ${print(statement.init)}`; + case "assign": + return `${pad}${print(statement.target)} = ${print(statement.value)}`; + case "if": { + const head = `${pad}if ${unwrapParens(print(statement.test))} {\n${printBody(statement.then, depth + 1)}\n${pad}}`; + return statement.otherwise.length === 0 + ? head + : `${head} else {\n${printBody(statement.otherwise, depth + 1)}\n${pad}}`; + } + case "switch": { + const cases = statement.cases + .map((entry) => `${pad}case ${entry.values.map((value) => literal(value)).join(", ")}:\n${printBody(entry.body, depth + 1)}`) + .join("\n"); + const fallback = + statement.otherwise === undefined ? "" : `\n${pad}default:\n${printBody(statement.otherwise, depth + 1)}`; + return `${pad}switch ${print(statement.subject)} {\n${cases}${fallback}\n${pad}}`; + } + case "for": { + const comparison = statement.step > 0n ? (statement.inclusive ? "<=" : "<") : statement.inclusive ? ">=" : ">"; + const update = statement.step === 1n ? `${statement.name}++` : statement.step === -1n ? `${statement.name}--` : `${statement.name} += ${statement.step}`; + return `${pad}for ${statement.name} := ${print(statement.from)}; ${statement.name} ${comparison} ${print(statement.to)}; ${update} {\n${printBody(statement.body, depth + 1)}\n${pad}}`; + } + case "forEach": { + // Go refuses to compile a declared local that nothing reads, and the subset allows a + // `for…of` whose body never touches its binding. The decision is made on the printed + // body rather than on the AST on purpose: a `raw` emission can carry a reference as + // text, invisible to a pass that walks nodes, and this search finds those too. It errs + // the safe way — a name that only looks used still gets its binding, which compiles. + const body = printBody(statement.body, depth + 1); + const reads = new RegExp(`\\b${statement.name}\\b`, "u").test(body); + // `for _ := range xs` is not the answer either: Go rejects a `:=` that binds nothing. + // A range loop that needs no element is written without the assignment at all. + const header = reads ? `for _, ${statement.name} := range ` : "for range "; + return `${pad}${header}${print(statement.iterable)} {\n${body}\n${pad}}`; + } + case "return": { + const values = [statement.value === undefined ? undefined : print(statement.value), ...(statement.extra ?? []).map(print)] + .filter((item): item is string => item !== undefined) + .join(", "); + return `${pad}return${values === "" ? "" : ` ${values}`}`; + } + case "throw": + return `${pad}return ZERO, &${statement.errorClass}{Message: ${statement.args.length === 0 ? '""' : print(statement.args[0]!)}}`; + case "break": + return `${pad}break`; + case "continue": + return `${pad}continue`; + case "expr": + return `${pad}${print(statement.expr)}`; + case "raw": + return `${pad}${statement.text}`; + default: { + const exhaustive: never = statement; + return exhaustive; + } + } +} + +function unwrapParens(text: string): string { + return text.startsWith("(") && text.endsWith(")") ? text.slice(1, -1) : text; +} + +export function printFunction(fn: TFunc): string { + const params = fn.params.map((param) => `${param.name} ${goType(param.type)}`).join(", "); + const returns = fn.fails.length > 0 ? `(${goType(fn.ret)}, error)` : goType(fn.ret); + const doc = fn.doc === undefined ? "" : `${fn.doc.split("\n").map((line) => `// ${line}`.trimEnd()).join("\n")}\n`; + const body = printBody(fn.body, 1).replaceAll("ZERO", goZero(fn.ret)); + return `${doc}func ${fn.name}(${params}) ${returns} {\n${body}\n}`; +} + +export function printRecord(record: TRecord): string { + const doc = record.doc === undefined ? "" : `// ${record.doc.split("\n")[0]}\n`; + const fields = record.fields.map((field) => `\t${field.name} ${goType(field.type)}`).join("\n"); + return `${doc}type ${record.name} struct {\n${fields}\n}`; +} + +/** A module path as an unexported Go identifier prefix: `is-valid-cpf` becomes `isValidCpf`. */ +function identifierPrefix(sourcePath: string): string { + const parts = sourcePath.split(/[^a-zA-Z0-9]+/u).filter((part) => part !== ""); + return parts + .map((part, index) => (index === 0 ? part : `${part[0]!.toUpperCase()}${part.slice(1)}`)) + .join(""); +} + +/** + * Lifts every `regexp.MustCompile` out of the function bodies and into a package level `var`. + * + * `regexp` has no compilation cache, so a `MustCompile` left inside a function recompiles the + * pattern from its source string on every call. For the mask-tolerant document patterns that is + * 79x the cost of the match itself, which made the generated validators slower than the handwritten + * package they replace. Compiling once at package initialisation is both the fix and what a Go + * author would have written. + * + * Every generated file shares one package, so the names carry the module they came from. + */ +function hoistPatterns(module: TModule, body: string): { body: string; declarations: string[] } { + const names = new Map(); + const prefix = identifierPrefix(module.sourcePath); + const hoisted = body.replaceAll(/regexp\.MustCompile\(("(?:[^"\\]|\\.)*")\)/gu, (_match, literal: string) => { + const existing = names.get(literal); + if (existing !== undefined) return existing; + const name = `${prefix}Pattern${names.size + 1}`; + names.set(literal, name); + return name; + }); + const declarations = [...names].map(([literal, name]) => `var ${name} = regexp.MustCompile(${literal})`); + return { body: hoisted, declarations }; +} + +export function printModule(module: TModule): string { + const printed = [ + ...module.records.map(printRecord), + ...module.constants.map( + (constant) => `var ${constant.name} = ${print(constant.value)}`, + ), + ...module.functions.map(printFunction), + ].join("\n\n"); + const patterns = hoistPatterns(module, printed); + const body = [...patterns.declarations, patterns.body].join("\n\n"); + const imports = new Set(module.requires.filter((name) => name !== "_support")); + for (const candidate of ["strings", "strconv", "regexp", "slices"]) { + if (new RegExp(`\\b${candidate}\\.`).test(body)) imports.add(candidate); + } + const importBlock = + imports.size === 0 + ? "" + : `import (\n${[...imports].sort().map((name) => `\t"${name}"`).join("\n")}\n)\n\n`; + return `${module.header}\n\npackage ${GO_CONFIG.packageName}\n\n${importBlock}${body}\n`; +} + +function importPath(): string { + // One package, so there are no cross-module imports to compute. + return ""; +} + +function supportModule(_program: CProgram, needs: SupportNeeds): { path: string; text: string } { + const parts = [ + "// Code generated by the logic engine. DO NOT EDIT.", + `// engine: ${ENGINE_VERSION}`, + "// source: support", + "", + `package ${GO_CONFIG.packageName}`, + "", + "import (", + '\t"slices"', + ")", + "", + "// ptr wraps a present value, which is how an Option is represented in Go.", + "func ptr[T any](value T) *T { return &value }", + "", + "// orElse answers the value, or the fallback when the option is absent.", + "func orElse[T any](value *T, fallback T) T {", + "\tif value == nil {", + "\t\treturn fallback", + "\t}", + "\treturn *value", + "}", + "", + "func codePoints(value string) []int {", + "\tpoints := make([]int, 0, len(value))", + "\tfor _, scalar := range value {", + "\t\tpoints = append(points, int(scalar))", + "\t}", + "\treturn points", + "}", + "", + "func fromCodePoints(points []int) string {", + "\tscalars := make([]rune, 0, len(points))", + "\tfor _, point := range points {", + "\t\tscalars = append(scalars, rune(point))", + "\t}", + "\treturn string(scalars)", + "}", + "", + "func asAscii(value string) *string {", + "\tfor _, scalar := range value {", + "\t\tif scalar >= 0x80 {", + "\t\t\treturn nil", + "\t\t}", + "\t}", + "\treturn &value", + "}", + "", + "func asDigits(value string) *string {", + "\tif value == \"\" {", + "\t\treturn nil", + "\t}", + "\tfor _, scalar := range value {", + "\t\tif scalar < '0' || scalar > '9' {", + "\t\t\treturn nil", + "\t\t}", + "\t}", + "\treturn &value", + "}", + "", + "func parseDigits(value string) *int {", + "\tif value == \"\" || len(value) > 18 {", + "\t\treturn nil", + "\t}", + "\ttotal := 0", + "\tfor _, scalar := range value {", + "\t\tif scalar < '0' || scalar > '9' {", + "\t\t\treturn nil", + "\t\t}", + "\t\ttotal = total*10 + int(scalar-'0')", + "\t}", + "\treturn &total", + "}", + "", + "func padStart(value string, length int, pad string) string {", + "\tscalars := []rune(value)", + "\tif len(scalars) >= length {", + "\t\treturn value", + "\t}", + "\tprefix := make([]rune, 0, length-len(scalars))", + "\tfor index := 0; index < length-len(scalars); index++ {", + "\t\tprefix = append(prefix, []rune(pad)...)", + "\t}", + "\treturn string(prefix) + value", + "}", + "", + "func codeAt(value string, index int) *int {", + "\tif index < 0 || index >= len(value) {", + "\t\treturn nil", + "\t}", + "\tpoint := int(value[index])", + "\treturn &point", + "}", + "", + "func charAt(value string, index int) *string {", + "\tif index < 0 || index >= len(value) {", + "\t\treturn nil", + "\t}", + "\tscalar := string(value[index])", + "\treturn &scalar", + "}", + "", + "func compareInts(left int, right int) int {", + "\tif left < right {", + "\t\treturn -1", + "\t}", + "\tif left > right {", + "\t\treturn 1", + "\t}", + "\treturn 0", + "}", + "", + "func absInt(value int) int {", + "\tif value < 0 {", + "\t\treturn -value", + "\t}", + "\treturn value", + "}", + "", + "func at[T any](values []T, index int) *T {", + "\tif index < 0 || index >= len(values) {", + "\t\treturn nil", + "\t}", + "\treturn &values[index]", + "}", + "", + "func sumInts(values []int) int {", + "\ttotal := 0", + "\tfor _, value := range values {", + "\t\ttotal += value", + "\t}", + "\treturn total", + "}", + "", + "func contains[T comparable](values []T, needle T) bool {", + "\treturn slices.Contains(values, needle)", + "}", + "", + "func indexOf[T comparable](values []T, needle T) int {", + "\treturn slices.Index(values, needle)", + "}", + "", + "func reversed[T any](values []T) []T {", + "\tout := make([]T, len(values))", + "\tfor index, value := range values {", + "\t\tout[len(values)-1-index] = value", + "\t}", + "\treturn out", + "}", + "", + "func sortedStable[T any](values []T, compare func(T, T) int) []T {", + "\tout := make([]T, len(values))", + "\tcopy(out, values)", + "\tslices.SortStableFunc(out, compare)", + "\treturn out", + "}", + "", + "func sortedStableBy[T any, K any](values []T, key func(T) K, compare func(K, K) int) []T {", + "\tout := make([]T, len(values))", + "\tcopy(out, values)", + "\tslices.SortStableFunc(out, func(left T, right T) int { return compare(key(left), key(right)) })", + "\treturn out", + "}", + "", + "func mapped[T any, R any](values []T, fn func(T) R) []R {", + "\tout := make([]R, 0, len(values))", + "\tfor _, value := range values {", + "\t\tout = append(out, fn(value))", + "\t}", + "\treturn out", + "}", + "", + "func filtered[T any](values []T, keep func(T) bool) []T {", + "\tout := make([]T, 0, len(values))", + "\tfor _, value := range values {", + "\t\tif keep(value) {", + "\t\t\tout = append(out, value)", + "\t\t}", + "\t}", + "\treturn out", + "}", + "", + "func anyOf[T any](values []T, test func(T) bool) bool {", + "\tfor _, value := range values {", + "\t\tif test(value) {", + "\t\t\treturn true", + "\t\t}", + "\t}", + "\treturn false", + "}", + "", + "func allOf[T any](values []T, test func(T) bool) bool {", + "\tfor _, value := range values {", + "\t\tif !test(value) {", + "\t\t\treturn false", + "\t\t}", + "\t}", + "\treturn true", + "}", + "", + "func found[T any](values []T, test func(T) bool) *T {", + "\tfor _, value := range values {", + "\t\tif test(value) {", + "\t\t\treturn ptr(value)", + "\t\t}", + "\t}", + "\treturn nil", + "}", + "", + "func dateFromEpochDays(days int) *int {", + "\tif days < -719162 || days > 2932896 {", + "\t\treturn nil", + "\t}", + "\treturn &days", + "}", + "", + "func dateAddDays(days int, shift int) *int {", + "\treturn dateFromEpochDays(days + shift)", + "}", + "", + ]; + if (needs.race) { + parts.push( + "// raceFirstSome runs idempotent tasks concurrently and takes the first one that answers.", + "// Cancellation is best effort: a losing goroutine may finish, and its answer is dropped.", + "func raceFirstSome[T any](tasks []func() *T) *T {", + "\tresults := make(chan *T, len(tasks))", + "\tfor _, task := range tasks {", + "\t\tgo func(run func() *T) {", + "\t\t\tresults <- run()", + "\t\t}(task)", + "\t}", + "", + "\tfor index := 0; index < len(tasks); index++ {", + "\t\tif answer := <-results; answer != nil {", + "\t\t\treturn answer", + "\t\t}", + "\t}", + "", + "\treturn nil", + "}", + "", + ); + } + if (needs.env) { + parts.push( + "// HttpHeader is one request or response header.", + "type HttpHeader struct {", + "\tName string", + "\tValue string", + "}", + "", + "// HttpRequest is a request handed to the environment.", + "type HttpRequest struct {", + "\tMethod string", + "\tUrl string", + "\tHeaders []HttpHeader", + "\tBody string", + "\tTimeoutMillis int", + "}", + "", + "// HttpResponse is a response from the environment.", + "type HttpResponse struct {", + "\tStatus int", + "\tHeaders []HttpHeader", + "\tBody string", + "}", + "", + "// Capabilities is everything the core needs from the outside world.", + "type Capabilities interface {", + "\t// A transport error or a timeout answers nil; a 4xx or 5xx status is a value.", + "\tRequest(request HttpRequest) *HttpResponse", + "\tNow() int", + "\tSleep(milliseconds int)", + "\tNextU32() int", + "}", + "", + ); + } + return { path: "support.go", text: parts.join("\n") }; +} + +function errorsModule(program: CProgram): { path: string; text: string } | undefined { + const declared = [...program.errors.values()]; + if (declared.length === 0) return undefined; + const lines = [ + "// Code generated by the logic engine. DO NOT EDIT.", + `// engine: ${ENGINE_VERSION}`, + "// source: errors", + "", + `package ${GO_CONFIG.packageName}`, + "", + 'import "errors"', + "", + "// ErrDomain is the root every domain error wraps, so errors.Is recognizes the family.", + 'var ErrDomain = errors.New("domain error")', + "", + ]; + for (const error of declared) { + lines.push( + `// ${error.name} ${error.doc?.split("\n")[0] ?? "is a domain error raised by the core."}`, + `type ${error.name} struct {`, + "\tMessage string", + "}", + "", + `func (e *${error.name}) Error() string { return e.Message }`, + "", + `func (e *${error.name}) Unwrap() error { return ErrDomain }`, + "", + ); + } + return { path: "errors.go", text: lines.join("\n") }; +} + + +/** The generated differential driver, plus the module file that makes the output buildable. */ +function driverFiles(_program: CProgram, entries: readonly DriverEntry[]): { path: string; text: string }[] { + // One argument, decoded from what `encoding/json` produced. It has to be recursive: a JSON + // array always arrives as `[]interface{}`, whatever the element type, so a `List` cannot + // simply be asserted to `[]int` — it has to be walked and converted element by element. The + // other three drivers get this for free from their languages' own decoding. + const value = (type: SemType, expr: string): string => { + if (type.kind === "Record") { + const definition = _program.records.get(type.name); + const fields = (definition?.fields ?? []).map((field) => { + const access = `${expr}.(map[string]interface{})[${JSON.stringify(field.name)}]`; + switch (field.type.kind) { + case "Bool": + return `${pascal(field.name)}: ${access}.(bool)`; + case "Int": + case "Decimal": + case "CivilDate": + return `${pascal(field.name)}: int(${access}.(float64))`; + default: + return `${pascal(field.name)}: ${access}.(string)`; + } + }); + return `core.${pascal(type.name)}{${fields.join(", ")}}`; + } + switch (type.kind) { + case "String": + case "Enum": + return `${expr}.(string)`; + case "Bool": + return `${expr}.(bool)`; + case "Float": + return `${expr}.(float64)`; + case "Int": + case "Decimal": + case "CivilDate": + case "Instant": + case "Duration": + return `int(${expr}.(float64))`; + case "List": { + const elem = goType(type.elem); + return [ + `func(raw interface{}) []${elem} {`, + "\titems := raw.([]interface{})", + `\tout := make([]${elem}, len(items))`, + "\tfor index, item := range items {", + `\t\tout[index] = ${value(type.elem, "item")}`, + "\t}", + "\treturn out", + `}(${expr})`, + ].join("\n"); + } + default: + return `${expr}.(${goType(type)})`; + } + }; + + const decode = (type: SemType, index: number): string => value(type, `args[${index}]`); + + const cases = entries.map((entry) => { + const call = `${entry.targetName}(${[ + ...entry.params.map((param, index) => decode(param, index)), + ...(entry.usesEnv ? ["environment"] : []), + ].join(", ")})`; + if (entry.fails.length > 0) { + return [ + `\tcase ${JSON.stringify(entry.coreName)}:`, + `\t\tvalue, err := core.${call}`, + "\t\tif err != nil {", + "\t\t\treturn nil, err", + "\t\t}", + "\t\treturn value, nil", + ].join("\n"); + } + return [`\tcase ${JSON.stringify(entry.coreName)}:`, `\t\treturn core.${call}, nil`].join("\n"); + }); + + const main = [ + "// Code generated by the logic engine. DO NOT EDIT.", + "// source: _driver", + "", + "package main", + "", + "import (", + '\t"bufio"', + '\t"encoding/json"', + '\t"fmt"', + '\t"os"', + ...(entries.some((entry) => entry.usesEnv) ? ['\t"time"'] : []), + "", + '\t"coreout"', + ")", + "", + "type request struct {", + '\tFn string `json:"fn"`', + '\tArgs []interface{} `json:"args"`', + "}", + "", + "func dispatch(name string, args []interface{}) (interface{}, error) {", + "\tswitch name {", + ...cases, + "\t}", + '\treturn nil, fmt.Errorf("unknown function %s", name)', + "}", + "", + ...(entries.some((entry) => entry.usesEnv) + ? [ + "// pcg32 is the reference PCG32: same constants and default seed as the interpreter's, so", + "// a draw matches the reference bit for bit. A fresh instance is built for every request,", + "// the same way the reference model starts a fresh interpreter -- and so a fresh generator", + "// -- per case.", + "type pcg32 struct {", + "\tstate uint64", + "\tincrement uint64", + "}", + "", + "func newPcg32(seed uint64) *pcg32 {", + "\tp := &pcg32{increment: 1442695040888963407}", + "\tp.next()", + "\tp.state += seed", + "\tp.next()", + "\treturn p", + "}", + "", + "func (p *pcg32) next() int {", + "\tprevious := p.state", + "\tp.state = previous*6364136223846793005 + p.increment", + "\txorshifted := uint32(((previous >> 18) ^ previous) >> 27)", + "\trotation := uint32(previous >> 59)", + "\treturn int((xorshifted >> rotation) | (xorshifted << ((-rotation) & 31)))", + "}", + "", + "// defaultSeed is the interpreter's own default: its constructor falls back to this seed", + "// whenever Capabilities.seed is left unset, which is how every conformance case runs it.", + "const defaultSeed uint64 = 0x853c49e6748fea9b", + "", + "// fakeCapabilities is the capability fake the differential harness drives: responses", + "// come from fixtures.json, a URL that is missing models a transport error, and the", + "// scripted latency is what decides a race.", + "type fixture struct {", + '\tStatus int `json:"status"`', + '\tBody string `json:"body"`', + '\tLatencyMillis int `json:"latencyMillis"`', + "}", + "", + "type fakeCapabilities struct {", + "\tfixtures map[string]fixture", + "\trandom *pcg32", + "}", + "", + "func (f fakeCapabilities) Request(request core.HttpRequest) *core.HttpResponse {", + "\tanswer, ok := f.fixtures[request.Url]", + "\tif !ok {", + "\t\treturn nil", + "\t}", + "\ttime.Sleep(time.Duration(answer.LatencyMillis) * time.Millisecond)", + "\treturn &core.HttpResponse{Status: answer.Status, Headers: []core.HttpHeader{}, Body: answer.Body}", + "}", + "", + "func (f fakeCapabilities) Now() int { return 0 }", + "", + "func (f fakeCapabilities) Sleep(milliseconds int) {", + "\ttime.Sleep(time.Duration(milliseconds) * time.Millisecond)", + "}", + "", + "func (f fakeCapabilities) NextU32() int { return f.random.next() }", + "", + "// A missing fixture file leaves every URL unanswered, which is a transport error.", + "var fixtures map[string]fixture = loadFixtures()", + "", + "var environment core.Capabilities", + "", + "// newEnvironment builds a fresh capability fake, so NextU32 starts from the same state", + "// the reference model's fresh interpreter starts from for every case.", + "func newEnvironment() core.Capabilities {", + "\treturn fakeCapabilities{fixtures: fixtures, random: newPcg32(defaultSeed)}", + "}", + "", + "func loadFixtures() map[string]fixture {", + "\tfixtures := map[string]fixture{}", + '\traw, err := os.ReadFile("fixtures.json")', + "\tif err == nil {", + "\t\tif err := json.Unmarshal(raw, &fixtures); err != nil {", + "\t\t\tpanic(err)", + "\t\t}", + "\t}", + "\treturn fixtures", + "}", + "", + ] + : []), + "func main() {", + + "\tscanner := bufio.NewScanner(os.Stdin)", + "\tscanner.Buffer(make([]byte, 1024*1024), 1024*1024)", + "", + "\tfor scanner.Scan() {", + '\t\tif scanner.Text() == "" {', + "\t\t\tcontinue", + "\t\t}", + "", + "\t\tvar parsed request", + "\t\tif err := json.Unmarshal(scanner.Bytes(), &parsed); err != nil {", + "\t\t\tpanic(err)", + "\t\t}", + "", + ...(entries.some((entry) => entry.usesEnv) + ? [ + "\t\t// A fresh environment per line: NextU32 starts from the same state the reference", + "\t\t// model's fresh interpreter starts from for every case.", + "\t\tenvironment = newEnvironment()", + "", + ] + : []), + "\t\tvalue, err := dispatch(parsed.Fn, parsed.Args)", + "\t\tif err != nil {", + '\t\t\tout, _ := json.Marshal(map[string]interface{}{"ok": false, "error": errorName(err)})', + "\t\t\tfmt.Println(string(out))", + "\t\t\tcontinue", + "\t\t}", + "", + '\t\tout, _ := json.Marshal(map[string]interface{}{"ok": true, "value": value})', + "\t\tfmt.Println(string(out))", + "\t}", + "}", + "", + "func errorName(err error) string {", + '\treturn fmt.Sprintf("%T", err)[len("*core."):]', + "}", + "", + ].join("\n"); + + return [ + { path: "cmd/driver/main.go", text: main }, + { path: "go.mod", text: `module coreout\n\ngo 1.21\n` }, + ]; +} + +export const GO_BACKEND: Backend = { + spec: GO_SPEC, + fileExtension: GO_CONFIG.fileExtension, + printModule, + importPath, + support: supportModule, + errorsModule, + renderType: goType, + comment: "//", + driver: driverFiles, +}; diff --git a/engine/src/targets/python/index.ts b/engine/src/targets/python/index.ts new file mode 100644 index 000000000..82b463436 --- /dev/null +++ b/engine/src/targets/python/index.ts @@ -0,0 +1,1210 @@ +/** + * The Python backend. + * + * Python 3.9, standard library only, `typing` annotations that pass pyright in strict mode. + * Comprehensions, builtins and early returns where they are the natural form; no code that + * imitates Go or Rust. Integers are Python `int`, which is already arbitrary precision, so a + * proven range never forces a different representation here. + */ + +import { dirname, relative as relativePath } from "node:path"; +import type { Backend, DriverEntry, SupportNeeds } from "../../backend/generate.ts"; +import { ENGINE_VERSION } from "../../backend/generate.ts"; +import type { TargetSpec } from "../../backend/lower.ts"; +import type { Candidate } from "../../backend/select.ts"; +import { LoweringTable, argIsAscii } from "../../backend/select.ts"; +import { asciiString, mapExprs } from "../../backend/tast.ts"; +import type { TExpr, TFunc, TModule, TRecord, TStmt } from "../../backend/tast.ts"; +import type { CProgram } from "../../core/ir.ts"; +import { printRegex } from "../../regex.ts"; +import type { SemType } from "../../types.ts"; +import type { Value } from "../../values.ts"; +import { TRIM_CODE_POINTS } from "../../intrinsics/index.ts"; + +export const PYTHON_CONFIG = { + baseline: "Python 3.9", + fileExtension: ".py", + dependencies: [] as string[], + formatter: "ruff format", + linters: ["ruff check", "pyright"], +}; + +export function pyType(type: SemType): string { + switch (type.kind) { + case "Bool": + return "bool"; + case "Int": + case "Decimal": + case "CivilDate": + case "Instant": + case "Duration": + return "int"; + case "Float": + return "float"; + case "String": + return "str"; + case "List": + return `List[${pyType(type.elem)}]`; + case "Option": + return `Optional[${pyType(type.inner)}]`; + case "Record": + return type.name; + case "Enum": + return `Literal[${type.members.map((member) => JSON.stringify(member)).join(", ")}]`; + case "Union": + return type.name; + case "Lambda": + return `Callable[[${type.params.map(pyType).join(", ")}], ${pyType(type.ret)}]`; + case "Void": + return "None"; + case "Never": + return "NoReturn"; + default: { + const exhaustive: never = type; + return exhaustive; + } + } +} + + +/** The body of a printed class, so a lowering can negate it. */ +function classBody(printed: string): string { + return printed.startsWith("[") && printed.endsWith("]") ? printed.slice(1, -1) : printed; +} + +const raw = (text: string): TExpr => ({ kind: "raw", text }); + +const NONE = raw("None"); + +/** + * The name `opt.orElse` binds its value to when it cannot fuse into a checked lowering. + * + * One name serves every use: the walrus is evaluated in the condition, before either branch reads + * it, so a nested `orElse` inside the value has already been read by the time the outer one + * rebinds, and two siblings are evaluated one after the other. + */ +const ORELSE_TEMP = "__value"; + +/** + * `value if test else None`, the shape every checked lowering takes in Python. + * + * It is a node rather than a string so `opt.orElse` can drop its default straight into the `else` + * branch. Printed on its own it is exactly the text it replaced. + */ +function checked(value: string, test: string): TExpr { + return { kind: "ternary", test: raw(test), then: raw(value), otherwise: NONE }; +} + +function binary(op: string): Candidate["emit"] { + return (args) => ({ kind: "binary", op, left: args[0]!, right: args[1]! }); +} + +const cheap = { alloc: "none", time: "constant" } as const; +const linear = { alloc: "none", time: "linear" } as const; +const allocating = { alloc: "one", time: "linear" } as const; +/** A pass that materializes the scalars of a string: correct everywhere, and the slowest option. */ +const scalarPass = { alloc: "many", time: "linear" } as const; + +function nonNegative(index: number) { + return (args: readonly SemType[]): boolean => { + const arg = args[index]; + return arg !== undefined && arg.kind === "Int" && arg.lo >= 0n; + }; +} + +/** Inlines a single-expression lambda, which is what makes a comprehension readable. */ +function comprehension(list: TExpr, lambda: TExpr, shape: "map" | "filter"): TExpr { + if (lambda.kind === "lambda" && lambda.body.length === 1 && lambda.body[0]!.kind === "return") { + const body = lambda.body[0]!.value!; + const name = lambda.params[0]!.name; + return shape === "map" + ? raw(`[${print(body)} for ${name} in ${print(list)}]`) + : raw(`[${name} for ${name} in ${print(list)} if ${print(body)}]`); + } + return shape === "map" + ? raw(`[__fn(__item) for __item in ${print(list)}]`.replace("__fn", print(lambda))) + : raw(`[__item for __item in ${print(list)} if ${print(lambda)}(__item)]`); +} + +function generatorOf(list: TExpr, lambda: TExpr): { name: string; test: string; list: string } { + if (lambda.kind === "lambda" && lambda.body.length === 1 && lambda.body[0]!.kind === "return") { + return { + name: lambda.params[0]!.name, + test: print(lambda.body[0]!.value!), + list: print(list), + }; + } + return { name: "__item", test: `${print(lambda)}(__item)`, list: print(list) }; +} + +export const PYTHON_CANDIDATES: readonly Candidate[] = [ + ...["add:+", "sub:-", "mul:*"].map((entry) => { + const [op, symbol] = entry.split(":") as [string, string]; + return { op: `int.${op}`, impl: "native" as const, cost: cheap, emit: binary(symbol) }; + }), + ...["add:+", "sub:-", "mul:*", "div:/"].map((entry) => { + const [op, symbol] = entry.split(":") as [string, string]; + return { op: `float.${op}`, impl: "native" as const, cost: cheap, emit: binary(symbol) }; + }), + { + op: "int.div", + impl: "native", + requires: (args) => nonNegative(0)(args) && nonNegative(1)(args), + because: "`//` floors, so it only matches truncated division when both operands are non-negative", + cost: cheap, + emit: binary("//"), + }, + { + // `trunc_div` (`_support.py`) is a Python-level function, and CPython's call overhead — a new + // frame, argument binding, a `return` — costs more than the arithmetic it wraps; a modulo or + // division whose operands are not provably non-negative pays that on every call, everywhere + // in the generated program (`group_thousands`' per-character loop is one such caller, and + // most of `formatCurrency`'s gap traced to it — `engine/docs/progress.md` §8). Printing the + // same formula inline removes the call. The tuple's job is single evaluation: `left`/`right` + // each appear once in the test (bound to `__td_a`/`__td_b`) and once more in whichever branch + // the ternary actually takes, never twice in the same executed path, so a non-trivial operand + // expression (not just a name) is still evaluated exactly once. + op: "int.div", + impl: "library", + cost: cheap, + emit: (args) => { + const left = print(args[0]!); + const right = print(args[1]!); + return raw( + `(-(abs(__td_a) // abs(__td_b)) if ((__td_a := ${left}), (__td_b := ${right}), (__td_a < 0) != (__td_b < 0))[2] else abs(__td_a) // abs(__td_b))`, + ); + }, + }, + { + op: "int.mod", + impl: "native", + requires: (args) => nonNegative(0)(args) && nonNegative(1)(args), + because: "`%` is floored in Python, so it only matches the Core's truncated remainder for non-negative operands", + cost: cheap, + emit: binary("%"), + }, + { + // Same reasoning as `int.div` above: `trunc_mod`'s own formula, inlined instead of called. + op: "int.mod", + impl: "library", + cost: cheap, + emit: (args) => { + const left = print(args[0]!); + const right = print(args[1]!); + return raw( + `(-(abs(__tm_a) % abs(__tm_b)) if ((__tm_a := ${left}), (__tm_b := ${right}), __tm_a < 0)[2] else abs(__tm_a) % abs(__tm_b))`, + ); + }, + }, + { op: "int.neg", impl: "native", cost: cheap, emit: (args) => raw(`-${print(args[0]!)}`) }, + { op: "float.neg", impl: "native", cost: cheap, emit: (args) => raw(`-${print(args[0]!)}`) }, + { op: "int.abs", impl: "native", cost: cheap, emit: (args) => raw(`abs(${print(args[0]!)})`) }, + { op: "int.min", impl: "native", cost: cheap, emit: (args) => raw(`min(${print(args[0]!)}, ${print(args[1]!)})`) }, + { op: "int.max", impl: "native", cost: cheap, emit: (args) => raw(`max(${print(args[0]!)}, ${print(args[1]!)})`) }, + ...["lt:<", "le:<=", "gt:>", "ge:>="].flatMap((entry) => { + const [op, symbol] = entry.split(":") as [string, string]; + return [ + { op: `int.${op}`, impl: "native" as const, cost: cheap, emit: binary(symbol) }, + { op: `float.${op}`, impl: "native" as const, cost: cheap, emit: binary(symbol) }, + ]; + }), + { op: "float.fromInt", impl: "native", cost: cheap, emit: (args) => raw(`float(${print(args[0]!)})`) }, + { op: "core.eq", impl: "native", cost: cheap, emit: binary("==") }, + + // A `binary` rather than text, so negating it prints `is not None` instead of `not … is None`. + { op: "opt.isNone", impl: "native", cost: cheap, emit: (args) => ({ kind: "binary", op: "is", left: args[0]!, right: NONE }) }, + { op: "opt.unwrap", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { op: "opt.some", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { + op: "opt.orElse", + impl: "native", + cost: cheap, + emit: (args, types) => { + const value = args[0]!; + // A checked lowering already answers `value if test else None`, so the default belongs + // in that `else` branch: the test runs once, and the result reads the way a Python + // author would have written it. It is only the same expression while the value itself + // can never be `None`, which an Option of an Option would break. + if (value.kind === "ternary" && value.otherwise === NONE && types[0]?.kind === "Option" && types[0].inner.kind !== "Option") { + return { kind: "ternary", test: value.test, then: value.then, otherwise: args[1]! }; + } + // Anything else has to be named before it can be tested and then answered, or it would + // be evaluated twice — for a call, that is the whole call run twice. One walrus binds + // it; the condition is evaluated first, so the name always holds this value by the time + // the branches read it, and a nested `orElse` rebinds it only after its own use. + if (value.kind === "name" || value.kind === "lit" || value.kind === "member") { + return raw(`(${print(value)} if ${print(value)} is not None else ${print(args[1]!)})`); + } + return raw(`(${ORELSE_TEMP} if (${ORELSE_TEMP} := ${print(value)}) is not None else ${print(args[1]!)})`); + }, + }, + + { + op: "str.len", + impl: "native", + because: "`len` counts code points in Python, which is the Core's definition", + cost: cheap, + emit: (args) => raw(`len(${print(args[0]!)})`), + }, + { op: "str.concat", impl: "native", cost: allocating, emit: binary("+") }, + { + op: "str.codeAt", + impl: "native", + requires: argIsAscii(0), + cost: cheap, + because: "indexing by scalar is O(1) and identical to the Core only for ASCII", + emit: (args) => raw(`ord(${print(args[0]!)}[${print(args[1]!)}])`), + }, + { + op: "str.charAt", + impl: "native", + requires: argIsAscii(0), + cost: cheap, + emit: (args) => raw(`${print(args[0]!)}[${print(args[1]!)}]`), + }, + { + op: "str.codeAtOpt", + impl: "native", + requires: argIsAscii(0), + cost: cheap, + emit: (args) => + checked( + `ord(${print(args[0]!)}[${print(args[1]!)}])`, + `0 <= ${print(args[1]!)} < len(${print(args[0]!)})`, + ), + }, + { + op: "str.charAtOpt", + impl: "native", + requires: argIsAscii(0), + cost: cheap, + emit: (args) => + checked(`${print(args[0]!)}[${print(args[1]!)}]`, `0 <= ${print(args[1]!)} < len(${print(args[0]!)})`), + }, + { + op: "str.slice", + impl: "native", + requires: argIsAscii(0), + cost: allocating, + emit: (args) => raw(`${print(args[0]!)}[${print(args[1]!)}:${print(args[2]!)}]`), + }, + { op: "str.indexOf", impl: "native", requires: argIsAscii(0), cost: linear, emit: (args) => raw(`${print(args[0]!)}.find(${print(args[1]!)})`) }, + { op: "str.contains", impl: "native", cost: linear, emit: (args) => raw(`(${print(args[1]!)} in ${print(args[0]!)})`) }, + { op: "str.startsWith", impl: "native", cost: linear, emit: (args) => raw(`${print(args[0]!)}.startswith(${print(args[1]!)})`) }, + { op: "str.endsWith", impl: "native", cost: linear, emit: (args) => raw(`${print(args[0]!)}.endswith(${print(args[1]!)})`) }, + { op: "str.repeat", impl: "native", cost: allocating, emit: binary("*") }, + { op: "str.padStart", impl: "native", cost: allocating, emit: (args) => raw(`${print(args[0]!)}.rjust(${print(args[1]!)}, ${print(args[2]!)})`) }, + { + op: "str.trim", + impl: "native", + because: "`strip()` uses Python's own whitespace set, so the 25 code points are passed explicitly", + cost: allocating, + emit: (args) => raw(`${print(args[0]!)}.strip(${asciiString(TRIM_CODE_POINTS.map((point) => String.fromCodePoint(point)).join(""))})`), + }, + { op: "str.asciiUpper", impl: "native", requires: argIsAscii(0), because: "`upper()` is only ASCII-equivalent on ASCII input", cost: allocating, emit: (args) => raw(`${print(args[0]!)}.upper()`) }, + { op: "str.asciiLower", impl: "native", requires: argIsAscii(0), because: "`lower()` is only ASCII-equivalent on ASCII input", cost: allocating, emit: (args) => raw(`${print(args[0]!)}.lower()`) }, + { + op: "str.asciiUpper", + impl: "native", + because: "one re.sub pass maps a-z and leaves every other scalar alone", + cost: allocating, + deps: ["re"], + emit: (args, _types, ctx) => { + ctx.require("re"); + return raw(`re.sub("[a-z]", lambda match: match.group().upper(), ${print(args[0]!)})`); + }, + }, + { + op: "str.asciiLower", + impl: "native", + because: "one re.sub pass maps A-Z and leaves every other scalar alone", + cost: allocating, + deps: ["re"], + emit: (args, _types, ctx) => { + ctx.require("re"); + return raw(`re.sub("[A-Z]", lambda match: match.group().lower(), ${print(args[0]!)})`); + }, + }, + { + op: "str.compare", + impl: "native", + because: "Python compares strings by code point, which is the Core's order", + cost: linear, + emit: (args) => raw(`(-1 if ${print(args[0]!)} < ${print(args[1]!)} else (1 if ${print(args[0]!)} > ${print(args[1]!)} else 0))`), + }, + { op: "str.codePoints", impl: "native", cost: allocating, emit: (args) => raw(`[ord(__c) for __c in ${print(args[0]!)}]`) }, + { op: "str.fromCodePoints", impl: "native", cost: allocating, emit: (args) => raw(`"".join(chr(__p) for __p in ${print(args[0]!)})`) }, + { op: "str.asAscii", impl: "native", cost: linear, emit: (args) => checked(print(args[0]!), `${print(args[0]!)}.isascii()`) }, + { + op: "str.asDigits", + impl: "native", + cost: linear, + emit: (args) => + checked( + print(args[0]!), + `(${print(args[0]!)} != "" and all("0" <= __c <= "9" for __c in ${print(args[0]!)}))`, + ), + }, + { op: "str.split", impl: "native", cost: allocating, emit: (args) => raw(`${print(args[0]!)}.split(${print(args[1]!)})`) }, + { op: "str.join", impl: "native", cost: allocating, emit: (args) => raw(`${print(args[1]!)}.join(${print(args[0]!)})`) }, + { op: "str.fromInt", impl: "native", cost: allocating, emit: (args) => raw(`str(${print(args[0]!)})`) }, + { + op: "str.parseInt", + impl: "native", + cost: linear, + emit: (args) => + checked( + `int(${print(args[0]!)})`, + `(${print(args[0]!)} != "" and len(${print(args[0]!)}) <= 18 and all("0" <= __c <= "9" for __c in ${print(args[0]!)}))`, + ), + }, + + { + op: "seq.at", + impl: "native", + cost: cheap, + emit: (args) => + checked(`${print(args[0]!)}[${print(args[1]!)}]`, `0 <= ${print(args[1]!)} < len(${print(args[0]!)})`), + }, + { + op: "str.asciiUpper", + impl: "portable", + // Building a scalar list costs far more than the host's own pass, which is why the cost + // class has to say so: selection ranks by cost before it ranks by implementation kind. + cost: scalarPass, + sourceFn: "std/strings::asciiUpperAll", + emit: (args, _types, ctx) => raw(`${ctx.nameOf("std/strings::asciiUpperAll")}(${print(args[0]!)})`), + }, + { + op: "str.asciiLower", + impl: "portable", + cost: scalarPass, + sourceFn: "std/strings::asciiLowerAll", + emit: (args, _types, ctx) => raw(`${ctx.nameOf("std/strings::asciiLowerAll")}(${print(args[0]!)})`), + }, + { op: "seq.len", impl: "native", cost: cheap, emit: (args) => raw(`len(${print(args[0]!)})`) }, + { op: "seq.get", impl: "native", cost: cheap, emit: (args) => raw(`${print(args[0]!)}[${print(args[1]!)}]`) }, + { op: "seq.push", impl: "native", cost: cheap, emit: (args) => raw(`${print(args[0]!)}.append(${print(args[1]!)})`) }, + { op: "seq.map", impl: "native", because: "a comprehension is the idiomatic map", cost: allocating, emit: (args) => comprehension(args[0]!, args[1]!, "map") }, + { op: "seq.filter", impl: "native", cost: allocating, emit: (args) => comprehension(args[0]!, args[1]!, "filter") }, + { + op: "seq.any", + impl: "native", + cost: linear, + emit: (args) => { + const gen = generatorOf(args[0]!, args[1]!); + return raw(`any(${gen.test} for ${gen.name} in ${gen.list})`); + }, + }, + { + op: "seq.all", + impl: "native", + cost: linear, + emit: (args) => { + const gen = generatorOf(args[0]!, args[1]!); + return raw(`all(${gen.test} for ${gen.name} in ${gen.list})`); + }, + }, + { + op: "seq.find", + impl: "native", + cost: linear, + emit: (args) => { + const gen = generatorOf(args[0]!, args[1]!); + return raw(`next((${gen.name} for ${gen.name} in ${gen.list} if ${gen.test}), None)`); + }, + }, + { op: "seq.sum", impl: "native", because: "`sum` is the idiomatic fold over +", cost: linear, emit: (args) => raw(`sum(${print(args[0]!)})`) }, + { op: "seq.indexOf", impl: "native", cost: linear, emit: (args) => raw(`(${print(args[0]!)}.index(${print(args[1]!)}) if ${print(args[1]!)} in ${print(args[0]!)} else -1)`) }, + { op: "seq.contains", impl: "native", cost: linear, emit: (args) => raw(`(${print(args[1]!)} in ${print(args[0]!)})`) }, + { op: "seq.concat", impl: "native", cost: allocating, emit: binary("+") }, + { op: "seq.slice", impl: "native", cost: allocating, emit: (args) => raw(`${print(args[0]!)}[${print(args[1]!)}:${print(args[2]!)}]`) }, + { op: "seq.reverse", impl: "native", cost: allocating, emit: (args) => raw(`list(reversed(${print(args[0]!)}))`) }, + { + op: "seq.sortStable", + impl: "library", + because: "`sorted` is stable; a comparator goes through functools.cmp_to_key", + cost: { alloc: "one", time: "nlogn" }, + deps: ["functools"], + emit: (args, _types, ctx) => { + ctx.require("functools"); + return raw(`sorted(${print(args[0]!)}, key=functools.cmp_to_key(${print(args[1]!)}))`); + }, + }, + { + op: "seq.sortStableBy", + impl: "native", + because: "`sorted(key=…)` is stable and avoids the comparator wrapper", + cost: { alloc: "one", time: "nlogn" }, + emit: (args) => raw(`sorted(${print(args[0]!)}, key=${print(args[1]!)})`), + }, + + { op: "dec.fromScaled", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { op: "dec.fromInt", impl: "native", cost: cheap, emit: (args, types) => raw(`${print(args[0]!)} * ${10 ** scaleOf(types[1])}`) }, + { op: "dec.add", impl: "native", cost: cheap, emit: binary("+") }, + { op: "dec.sub", impl: "native", cost: cheap, emit: binary("-") }, + { op: "dec.mul", impl: "native", cost: cheap, emit: binary("*") }, + { op: "dec.compare", impl: "native", cost: cheap, emit: (args) => raw(`(-1 if ${print(args[0]!)} < ${print(args[1]!)} else (1 if ${print(args[0]!)} > ${print(args[1]!)} else 0))`) }, + { op: "dec.isNegative", impl: "native", cost: cheap, emit: (args) => raw(`${print(args[0]!)} < 0`) }, + { op: "dec.abs", impl: "native", cost: cheap, emit: (args) => raw(`abs(${print(args[0]!)})`) }, + { op: "dec.unscaled", impl: "native", cost: cheap, emit: (args) => args[0]! }, + + { + op: "date.clampEpochDays", + impl: "native", + cost: cheap, + emit: (args) => raw(`min(max(${print(args[0]!)}, -719162), 2932896)`), + }, + { op: "date.toEpochDays", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { + op: "date.fromEpochDays", + impl: "native", + cost: cheap, + emit: (args) => checked(print(args[0]!), `-719162 <= ${print(args[0]!)} <= 2932896`), + }, + { + op: "date.fromYmd", + impl: "portable", + cost: linear, + sourceFn: "std/date::ymdToDays", + emit: (args, _types, ctx) => raw(`${ctx.nameOf("std/date::ymdToDays")}(${args.map(print).join(", ")})`), + }, + { op: "date.year", impl: "portable", cost: cheap, sourceFn: "std/date::yearFromDays", emit: (args, _types, ctx) => raw(`${ctx.nameOf("std/date::yearFromDays")}(${print(args[0]!)})`) }, + { op: "date.month", impl: "portable", cost: cheap, sourceFn: "std/date::monthFromDays", emit: (args, _types, ctx) => raw(`${ctx.nameOf("std/date::monthFromDays")}(${print(args[0]!)})`) }, + { op: "date.day", impl: "portable", cost: cheap, sourceFn: "std/date::dayFromDays", emit: (args, _types, ctx) => raw(`${ctx.nameOf("std/date::dayFromDays")}(${print(args[0]!)})`) }, + { + op: "date.addDays", + impl: "native", + cost: cheap, + emit: (args) => + checked( + `${print(args[0]!)} + ${print(args[1]!)}`, + `-719162 <= ${print(args[0]!)} + ${print(args[1]!)} <= 2932896`, + ), + }, + { op: "date.diffDays", impl: "native", cost: cheap, emit: binary("-") }, + { op: "date.compare", impl: "native", cost: cheap, emit: (args) => raw(`(-1 if ${print(args[0]!)} < ${print(args[1]!)} else (1 if ${print(args[0]!)} > ${print(args[1]!)} else 0))`) }, + { op: "date.dayOfWeek", impl: "native", cost: cheap, emit: (args) => raw(`((${print(args[0]!)} + 3) % 7) + 1`) }, + { op: "date.isLeapYear", impl: "native", cost: cheap, emit: (args) => raw(`((${print(args[0]!)} % 4 == 0 and ${print(args[0]!)} % 100 != 0) or ${print(args[0]!)} % 400 == 0)`) }, + + { + op: "re.retain", + impl: "native", + because: "re.sub with the negated class is one pass", + cost: allocating, + deps: ["re"], + emit: (args, _types, ctx) => { + ctx.require("re"); + const pattern = ctx.regex === undefined ? "" : printRegex(ctx.regex.node, "python"); + return raw(`re.sub(${asciiString(`[^${classBody(pattern)}]`)}, "", ${print(args[0]!)})`); + }, + }, + { + op: "re.test", + impl: "native", + because: "`fullmatch` anchors the whole string, and the normalized pattern uses explicit classes", + cost: linear, + deps: ["re"], + emit: (args, _types, ctx) => { + ctx.require("re"); + const pattern = ctx.regex === undefined ? "" : printRegex(ctx.regex.node, "python"); + return raw(`(re.fullmatch(${pythonPattern(pattern)}, ${print(args[0]!)}) is not None)`); + }, + }, + + { + op: "http.request", + impl: "native", + cost: { alloc: "many", time: "linear" }, + emit: (args, _types, ctx) => raw(`${print(ctx.env())}.request(${print(args[0]!)})`), + }, + { op: "clock.now", impl: "native", cost: cheap, emit: (_args, _types, ctx) => raw(`${print(ctx.env())}.now()`) }, + { op: "clock.sleep", impl: "native", cost: cheap, emit: (args, _types, ctx) => raw(`${print(ctx.env())}.sleep(${print(args[0]!)})`) }, + { op: "clock.millis", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { op: "clock.durationMillis", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { op: "clock.elapsed", impl: "native", cost: cheap, emit: (args) => raw(`max(0, ${print(args[1]!)} - ${print(args[0]!)})`) }, + { op: "random.nextU32", impl: "native", cost: cheap, emit: (_args, _types, ctx) => raw(`${print(ctx.env())}.next_u32()`) }, + { + op: "task.race", + impl: "library", + because: "a ThreadPoolExecutor is the standard library's way to run idempotent requests concurrently", + cost: { alloc: "many", time: "linear" }, + deps: ["_support"], + emit: (args, _types, ctx) => { + ctx.require("_support"); + return raw(`race_first_some(${print(args[0]!)})`); + }, + }, +]; + +function scaleOf(type: SemType | undefined): number { + return type !== undefined && type.kind === "Int" ? Number(type.lo) : 0; +} + +/** + * The pattern is emitted as an ordinary (escaped) Python string, not a raw one: the normalized + * form already contains backslash escapes, and JSON escaping turns each of them into the two + * characters the `re` module expects to read. + */ +function pythonPattern(pattern: string): string { + return asciiString(pattern); +} + +export const PYTHON_SPEC: TargetSpec = { + name: "python", + table: new LoweringTable(PYTHON_CANDIDATES), + naming: { + func: (name) => snake(name), + value: (name) => snake(name), + field: (name) => snake(name), + type: (name) => name, + module: (path) => `${path.split("/").map(snake).join("/")}.py`, + }, + // A fold is a loop in Python: the comprehension forms cover map and filter, and + // `functools.reduce` is not idiomatic. + loopCombinators: new Set(["seq.fold"]), + statementTernary: false, + errorsAsValues: false, + asyncColouring: false, + envType: { kind: "Record", name: "Capabilities" }, +}; + +function snake(name: string): string { + return name + .replace(/([a-z0-9])([A-Z])/g, "$1_$2") + .replaceAll("-", "_") + .toLowerCase(); +} + +/* ------------------------------------------------------------------ * + * Printer + * ------------------------------------------------------------------ */ + +export function print(expr: TExpr): string { + switch (expr.kind) { + case "lit": + return literal(expr.value); + case "name": + return expr.name; + case "raw": + return expr.text; + case "call": + return `${print(expr.callee)}(${expr.args.map(print).join(", ")})`; + case "method": + return `${print(expr.target)}.${snake(expr.name)}(${expr.args.map(print).join(", ")})`; + case "member": + return `${print(expr.target)}.${expr.name}`; + case "index": + return `${print(expr.target)}[${print(expr.index)}]`; + case "binary": + return `(${print(expr.left)} ${pythonOperator(expr.op)} ${print(expr.right)})`; + case "unary": { + if (expr.op === "!" && expr.operand.kind === "binary" && expr.operand.op === "==") { + return `(${print(expr.operand.left)} != ${print(expr.operand.right)})`; + } + if (expr.op === "!" && expr.operand.kind === "binary" && expr.operand.op === "is") { + return `(${print(expr.operand.left)} is not ${print(expr.operand.right)})`; + } + return expr.op === "!" ? `(not ${print(expr.operand)})` : `${expr.op}${print(expr.operand)}`; + } + case "ternary": + return `(${print(expr.then)} if ${print(expr.test)} else ${print(expr.otherwise)})`; + case "list": + return `[${expr.items.map(print).join(", ")}]`; + case "record": + return `${expr.typeName}(${expr.fields.map((field) => `${snake(field.name)}=${print(field.value)}`).join(", ")})`; + case "lambda": { + const params = expr.params.map((param) => param.name).join(", "); + if (expr.body.length === 1 && expr.body[0]!.kind === "return" && expr.body[0]!.value !== undefined) { + return `(lambda ${params}: ${print(expr.body[0]!.value!)})`; + } + return `(lambda ${params}: None)`; + } + case "none": + case "zero": + return "None"; + case "some": + return print(expr.inner); + default: { + const exhaustive: never = expr; + return exhaustive; + } + } +} + +function pythonOperator(op: string): string { + switch (op) { + case "&&": + return "and"; + case "||": + return "or"; + case "===": + return "=="; + case "!==": + return "!="; + default: + return op; + } +} + +function literal(value: Value): string { + if (typeof value === "bigint") return value.toString(); + if (typeof value === "string") return asciiString(value); + if (typeof value === "boolean") return value ? "True" : "False"; + if (typeof value === "number") return String(value); + if (Array.isArray(value)) return `[${value.map(literal).join(", ")}]`; + return "None"; +} + +function printBody(body: readonly TStmt[], depth: number): string { + if (body.length === 0) return `${" ".repeat(depth)}pass`; + return body.map((statement) => printStmt(statement, depth)).join("\n"); +} + +function printStmt(statement: TStmt, depth: number): string { + const pad = " ".repeat(depth); + switch (statement.kind) { + case "let": + return `${pad}${statement.name}: ${pyType(statement.type)} = ${print(statement.init)}`; + case "multiLet": + return `${pad}${statement.names.join(", ")} = ${print(statement.init)}`; + case "assign": + return `${pad}${print(statement.target)} = ${print(statement.value)}`; + case "if": { + const head = `${pad}if ${print(statement.test)}:\n${printBody(statement.then, depth + 1)}`; + return statement.otherwise.length === 0 + ? head + : `${head}\n${pad}else:\n${printBody(statement.otherwise, depth + 1)}`; + } + case "switch": { + const branches = statement.cases.map((entry, index) => { + const test = entry.values + .map((value) => `${print(statement.subject)} == ${literal(value)}`) + .join(" or "); + return `${pad}${index === 0 ? "if" : "elif"} ${test}:\n${printBody(entry.body, depth + 1)}`; + }); + if (statement.otherwise !== undefined) { + branches.push(`${pad}else:\n${printBody(statement.otherwise, depth + 1)}`); + } + return branches.join("\n"); + } + case "for": { + const step = statement.step; + const to = print(statement.to); + const bound = statement.inclusive + ? step > 0n + ? `${to} + 1` + : `${to} - 1` + : to; + const stepArg = step === 1n ? "" : `, ${step}`; + return `${pad}for ${statement.name} in range(${print(statement.from)}, ${bound}${stepArg}):\n${printBody(statement.body, depth + 1)}`; + } + case "forEach": + return `${pad}for ${statement.name} in ${print(statement.iterable)}:\n${printBody(statement.body, depth + 1)}`; + case "return": + return statement.value === undefined ? `${pad}return` : `${pad}return ${print(statement.value)}`; + case "throw": + return `${pad}raise ${statement.errorClass}(${statement.args.map(print).join(", ")})`; + case "break": + return `${pad}break`; + case "continue": + return `${pad}continue`; + case "expr": + return `${pad}${print(statement.expr)}`; + case "raw": + return `${pad}${statement.text}`; + default: { + const exhaustive: never = statement; + return exhaustive; + } + } +} + +/** + * A docstring whose backslashes survive: a plain Python string reads `\\u` as an escape. + * + * A doc that spans lines keeps all of them, indented to the body with the closing quotes on their + * own line, the shape PEP 257 describes. Keeping only the first line used to cut a sentence in + * half, since the source wraps its prose. + */ +function docstring(text: string, indent: string): string { + const lines = text.split("\n").map((line) => line.replaceAll("\\", "\\\\").replaceAll(TRIPLE_QUOTE, "'''")); + if (lines.length === 1) return `${TRIPLE_QUOTE}${lines[0]}${TRIPLE_QUOTE}`; + const body = lines.map((line) => (line === "" ? "" : `${indent}${line}`)).join("\n").trimStart(); + return `${TRIPLE_QUOTE}${body}\n${indent}${TRIPLE_QUOTE}`; +} + +const TRIPLE_QUOTE = '"'.repeat(3); + +export function printFunction(fn: TFunc): string { + const params = fn.params.map((param) => `${param.name}: ${pyType(param.type)}`).join(", "); + const doc = fn.doc === undefined ? "" : ` ${docstring(fn.doc, " ")}\n`; + return `def ${fn.name}(${params}) -> ${pyType(fn.ret)}:\n${doc}${printBody(fn.body, 1)}`; +} + +export function printRecord(record: TRecord): string { + const fields = record.fields.map((field) => ` ${snake(field.name)}: ${pyType(field.type)}`).join("\n"); + const doc = record.doc === undefined ? "" : ` ${docstring(record.doc, " ")}\n`; + return `@dataclass(frozen=True)\nclass ${record.name}:\n${doc}${fields}`; +} + +/** A module path as a Python identifier fragment: `lib/cnpj` becomes `LIB_CNPJ`. */ +function identifierPrefix(sourcePath: string): string { + return sourcePath.split(/[^a-zA-Z0-9]+/u).filter((part) => part !== "").join("_").toUpperCase(); +} + +/** + * Compiles each pattern once, at module level, instead of on every call. + * + * `re.fullmatch(pattern, value)` and `re.sub(pattern, …)` look the pattern up in `re`'s cache by + * its source string on every call, and these patterns are long. Measured, that lookup is about + * 50 ms per 200 000 calls for each of the two, against a compiled pattern's method call — roughly + * 15% of a validator. The pattern is known when the code is generated, so it is compiled then, + * which is also what a Python author would write. + * + * The names carry the module so a reader can tell where a pattern came from. + */ +function hoistPatterns(module: TModule, body: string): { body: string; declarations: string[] } { + const names = new Map(); + const prefix = identifierPrefix(module.sourcePath); + const hoisted = body.replaceAll( + /\bre\.(fullmatch|sub)\((("(?:[^"\\]|\\.)*")|('(?:[^'\\]|\\.)*'))(, )/gu, + (_match, method: string, literal: string, _double: string, _single: string, tail: string) => { + const existing = names.get(literal); + const name = existing ?? `_${prefix}_PATTERN_${names.size + 1}`; + if (existing === undefined) names.set(literal, name); + // `re.sub(pattern, repl, value)` becomes `pattern.sub(repl, value)`: the pattern stops + // being an argument, and the separator that followed it goes with it. + void tail; + return `${name}.${method}(`; + }, + ); + const declarations = [...names].map(([literal, name]) => `${name} = re.compile(${literal})`); + return { body: hoisted, declarations }; +} + +/** + * Marks every function its source module never exported as Python's own notion of private: a + * leading underscore, on the declaration and on every call site that names it. Every such + * function is called only from within its own module (`docs/semantics.md`'s effects section + * surveys the whole project), so a module-local rename is enough — nothing outside ever needs the + * unprefixed name resolved. + */ +function applyPrivacy(module: TModule): TModule { + const renamed = new Map(); + for (const fn of module.functions) { + if (!fn.moduleExported) renamed.set(fn.name, `_${fn.name}`); + } + if (renamed.size === 0) return module; + const rename = (expr: TExpr): TExpr => { + if (expr.kind === "name" && renamed.has(expr.name)) return { ...expr, name: renamed.get(expr.name)! }; + // A lowering-table candidate (`task.race`'s, among others) can pre-render a fragment of + // text at lowering time, baking in whatever name the callee had then — `mapExprs` cannot + // see inside it the way it sees a "call" node's own callee, so the same rename repeats here + // as a word-boundary substitution, the technique `hoistPatterns` already uses in this file. + if (expr.kind === "raw") { + let text = expr.text; + for (const [from, to] of renamed) text = text.replaceAll(new RegExp(`\\b${from}\\b`, "gu"), to); + return text === expr.text ? expr : { ...expr, text }; + } + return expr; + }; + return { + ...module, + functions: module.functions.map((fn) => ({ + ...fn, + name: renamed.get(fn.name) ?? fn.name, + body: mapExprs(fn.body, rename), + })), + }; +} + +export function printModule(rawModule: TModule): string { + const module = applyPrivacy(rawModule); + const typingNames = new Set(); + // Module level constants are upper cased, the way Python names a constant, so every reference + // to one has to be upper cased too. Driven by the module's own constants rather than by a + // pattern that guesses at their names, which silently missed any name shaped differently. + const text = module.constants.reduce( + (body, constant) => + body.replaceAll(new RegExp(`\\b${constant.name}\\b`, "gu"), constant.name.toUpperCase()), + module.functions.map(printFunction).join("\n\n\n"), + ); + const patterns = hoistPatterns(module, text); + const records = module.records.map(printRecord).join("\n\n\n"); + const constants = module.constants + .map((constant) => `${constant.name.toUpperCase()}: ${pyType(constant.type)} = ${print(constant.value)}`) + .join("\n"); + // Scanned over everything that is annotated, constants included: a hoisted table is declared + // `List[int]` at module level, and which module it lands in moves with whichever function still + // reads it, so a module can acquire one without any of its functions mentioning a `List` at all. + for (const name of ["List", "Optional", "Literal", "Callable", "NoReturn"]) { + if (new RegExp(`\\b${name}\\[`).test(`${text}${records}${constants}`) || text.includes(`-> ${name}`)) { + typingNames.add(name); + } + } + const parts: string[] = [module.header, ""]; + if (typingNames.size > 0) parts.push(`from typing import ${[...typingNames].sort().join(", ")}`); + if (records !== "") parts.push("from dataclasses import dataclass"); + for (const name of module.requires) { + if (name !== "_support") { + parts.push(`import ${name}`); + continue; + } + // `_support` lives at the package root, so a nested module walks back up to it. + const dots = ".".repeat(module.sourcePath.split("/").length); + const used = ["race_first_some"].filter((helper) => new RegExp(`\\b${helper}\\(`).test(text)); + if (used.length > 0) parts.push(`from ${dots}_support import ${used.join(", ")}`); + } + for (const item of module.imports) { + if (item.from === "SUPPORT") { + const dots = ".".repeat(module.sourcePath.split("/").length); + parts.push(`from ${dots}_support import ${item.names.join(", ")}`); + continue; + } + // Classes keep their name; functions are snake cased like everything else. + parts.push( + `from ${item.from} import ${item.names.map((name) => (/^[A-Z]/.test(name) ? name : snake(name))).join(", ")}`, + ); + } + parts.push(""); + // `__all__` names the module's public surface explicitly, on top of the leading underscore + // `applyPrivacy` already gave every unexported function — the two devices Python convention + // pairs for exactly this, so `from module import *` matches what the source module exported. + const publicFunctions = module.functions.filter((fn) => fn.moduleExported).map((fn) => fn.name); + if (publicFunctions.length > 0) { + parts.push(`__all__ = [${publicFunctions.map((name) => JSON.stringify(name)).join(", ")}]`, ""); + } + if (records !== "") parts.push(records, ""); + if (constants !== "") parts.push(...constants.split("\n").flatMap((line) => [line, ""])); + for (const declaration of patterns.declarations) parts.push(declaration, ""); + parts.push(patterns.body); + return `${parts.join("\n").trimEnd()}\n`; +} + +function importPath(from: string, to: string): string { + const rel = relativePath(dirname(from) === "." ? "" : dirname(from), to).split("\\").join("/"); + const dots = rel.startsWith("..") ? ".." : "."; + return `${dots}${rel.replaceAll("../", "").split("/").map(snake).join(".")}`; +} + +function supportModule(_program: CProgram, needs: SupportNeeds): { path: string; text: string } | undefined { + const parts = [ + "# Code generated by the logic engine. DO NOT EDIT.", + `# engine: ${ENGINE_VERSION}`, + "# source: _support", + "", + "from dataclasses import dataclass", + "from typing import Callable, List, Optional, Sequence, TypeVar", + "", + "T = TypeVar(\"T\")", + "", + ]; + if (needs.race) { + parts.push( + "", + "def race_first_some(tasks: Sequence[Callable[[], Optional[T]]]) -> Optional[T]:", + " from concurrent.futures import ThreadPoolExecutor, as_completed", + "", + " with ThreadPoolExecutor(max_workers=max(1, len(tasks))) as pool:", + " futures = [pool.submit(task) for task in tasks]", + "", + " for future in as_completed(futures):", + " try:", + " return future.result()", + " except Exception:", + " continue", + "", + " return None", + "", + ); + } + if (needs.env) { + parts.push( + "", + "@dataclass(frozen=True)", + "class HttpHeader:", + " name: str", + " value: str", + "", + "", + "@dataclass(frozen=True)", + "class HttpRequest:", + " method: str", + " url: str", + " headers: List[HttpHeader]", + " body: str", + " timeout_millis: int", + "", + "", + "@dataclass(frozen=True)", + "class HttpResponse:", + " status: int", + " headers: List[HttpHeader]", + " body: str", + "", + "", + "class Capabilities:", + ' """The default environment, built from the standard library only."""', + "", + " def request(self, request: HttpRequest) -> Optional[HttpResponse]:", + " import json", + " import urllib.error", + " import urllib.request", + "", + " payload = None if request.body == \"\" else request.body.encode()", + " parsed = urllib.request.Request(request.url, data=payload, method=request.method)", + "", + " for header in request.headers:", + " parsed.add_header(header.name, header.value)", + "", + " try:", + " with urllib.request.urlopen(parsed, timeout=request.timeout_millis / 1000) as response:", + " return HttpResponse(status=response.status, headers=[], body=response.read().decode())", + " except urllib.error.HTTPError as error:", + " return HttpResponse(status=error.code, headers=[], body=error.read().decode())", + " except Exception:", + " return None", + "", + " def now(self) -> int:", + " import time", + "", + " return int(time.time() * 1000)", + "", + " def sleep(self, milliseconds: int) -> None:", + " import time", + "", + " time.sleep(milliseconds / 1000)", + "", + " def next_u32(self) -> int:", + " import random", + "", + " # Not cryptographically secure, deliberately: the utilities that draw are", + " # generating example documents, and that is what the published package", + " # documents doing. A caller who needs unpredictability passes its own", + " # capability, the way the conformance harness passes a seeded one.", + " return random.getrandbits(32)", + "", + "", + "# The platform default, built once at import time rather than per call — every public", + "# wrapper (docs/decisions/0011-public-entry-points-vs-capabilities.md) shares this one", + "# instance, the same way a caller who builds their own environment would share it.", + "DEFAULT_CAPABILITIES = Capabilities()", + "", + ); + } + return { path: "_support.py", text: parts.join("\n") }; +} + +function errorsModule(program: CProgram): { path: string; text: string } | undefined { + const declared = [...program.errors.values()]; + if (declared.length === 0) return undefined; + const lines = [ + "# Code generated by the logic engine. DO NOT EDIT.", + `# engine: ${ENGINE_VERSION}`, + "# source: errors", + "", + "", + "class DomainError(Exception):", + ' """The root of every domain error the core raises."""', + "", + ]; + for (const error of declared) { + lines.push( + "", + `class ${error.name}(${error.base ?? "DomainError"}):`, + ` ${docstring(error.doc ?? error.name, " ")}`, + "", + ); + } + return { path: "errors.py", text: lines.join("\n") }; +} + + +/** The generated differential driver: one JSON line in, one JSON line out. */ +function driverFiles(_program: CProgram, entries: readonly DriverEntry[]): { path: string; text: string }[] { + const lines = [ + "# Code generated by the logic engine. DO NOT EDIT.", + "# source: _driver", + "", + "import dataclasses", + "import json", + "import sys", + "", + "", + "def encode(value: object) -> object:", + ' """A generated record is a dataclass, which json.dumps does not know."""', + " if dataclasses.is_dataclass(value):", + " return dataclasses.asdict(value)", + " raise TypeError(value)", + "", + ]; + const byModule = new Map(); + for (const entry of entries) { + const module = `.${entry.modulePath.replace(/\.py$/, "").split("/").join(".")}`; + const names = byModule.get(module) ?? []; + names.push(entry.targetName); + byModule.set(module, names); + } + for (const [module, names] of byModule) { + const records = entries + .filter((entry) => `.${entry.modulePath.replace(/\.py$/, "").split("/").join(".")}` === module) + .flatMap((entry) => entry.params.filter((param) => param.kind === "Record").map((param) => (param as { name: string }).name)); + lines.push(`from ${module} import ${[...new Set([...names, ...records])].sort().join(", ")}`); + } + const needsEnv = entries.some((entry) => entry.usesEnv); + if (needsEnv) { + lines.push( + "import os", + "import time", + "from typing import Optional", + "from ._support import Capabilities, HttpRequest, HttpResponse", + "", + "", + "# The reference PCG32: same constants and default seed as the interpreter's, so a draw", + "# matches the reference bit for bit. A fresh instance is built for every request, the same", + "# way the reference model starts a fresh interpreter, and so a fresh generator, per case.", + "class Pcg32:", + "", + " MASK64 = (1 << 64) - 1", + " MASK32 = (1 << 32) - 1", + " INCREMENT = 1442695040888963407", + "", + " def __init__(self, seed: int) -> None:", + " self.state = 0", + " self.next_u32()", + " self.state = (self.state + seed) & self.MASK64", + " self.next_u32()", + "", + " def next_u32(self) -> int:", + " previous = self.state", + " self.state = (previous * 6364136223846793005 + self.INCREMENT) & self.MASK64", + " xorshifted = (((previous >> 18) ^ previous) >> 27) & self.MASK32", + " rotation = previous >> 59", + " return ((xorshifted >> rotation) | (xorshifted << ((-rotation) & 31))) & self.MASK32", + "", + "", + "# The interpreter's own default: its constructor falls back to this seed whenever", + "# `Capabilities.seed` is left unset, which is how every conformance case runs it.", + "DEFAULT_SEED = 0x853C49E6748FEA9B", + "", + "", + "class FakeCapabilities:", + ' """The capability fake the differential harness drives: responses come from fixtures.json."""', + "", + " def __init__(self, fixtures: dict) -> None:", + " self.fixtures = fixtures", + " self.random = Pcg32(DEFAULT_SEED)", + "", + " def request(self, request: HttpRequest) -> Optional[HttpResponse]:", + " fixture = self.fixtures.get(request.url)", + "", + " if fixture is None:", + " return None", + "", + ' time.sleep(fixture.get("latencyMillis", 0) / 1000)', + "", + ' return HttpResponse(status=fixture["status"], headers=[], body=fixture["body"])', + "", + " def now(self) -> int:", + " return 0", + "", + " def sleep(self, milliseconds: int) -> None:", + " time.sleep(milliseconds / 1000)", + "", + " def next_u32(self) -> int:", + " return self.random.next_u32()", + "", + "", + 'FIXTURE_PATH = os.path.join(os.path.dirname(__file__), "fixtures.json")', + "", + "if os.path.exists(FIXTURE_PATH):", + " with open(FIXTURE_PATH) as handle:", + " FIXTURES = json.load(handle)", + "else:", + " FIXTURES = None", + ); + } + lines.push( + "", + "HANDLERS = {", + ...entries.map((entry) => { + const args = entry.params.map((param, index) => { + if (param.kind !== "Record") return `args[${index}]`; + const definition = _program.records.get(param.name); + const fields = (definition?.fields ?? []) + .map((field) => `${snake(field.name)}=args[${index}][${JSON.stringify(field.name)}]`) + .join(", "); + return `${param.name}(${fields})`; + }); + if (entry.usesEnv) args.push("ENVIRONMENT"); + return ` ${JSON.stringify(entry.coreName)}: lambda args: ${entry.targetName}(${args.join(", ")}),`; + }), + "}", + "", + "", + "for line in sys.stdin:", + ' if line.strip() == "":', + " continue", + "", + " request = json.loads(line)", + "", + ...(needsEnv + ? [ + " # A fresh environment per line: next_u32 starts from the same state the reference", + " # model's fresh interpreter starts from for every case.", + " ENVIRONMENT = FakeCapabilities(FIXTURES) if FIXTURES is not None else Capabilities()", + "", + ] + : []), + " try:", + ' value = HANDLERS[request["fn"]](request["args"])', + ' print(json.dumps({"ok": True, "value": value}, default=encode))', + " except Exception as error:", + ' print(json.dumps({"ok": False, "error": type(error).__name__}))', + "", + " sys.stdout.flush()", + "", + ); + const packages = new Set([""]); + for (const fn of _program.functions.values()) { + const parts = fn.module.split("/"); + for (let index = 1; index < parts.length; index++) { + packages.add(parts.slice(0, index).map(snake).join("/")); + } + } + return [ + { path: "_driver.py", text: lines.join("\n") }, + ...[...packages].map((directory) => ({ + path: directory === "" ? "__init__.py" : `${directory}/__init__.py`, + text: "", + })), + ]; +} + +export const PYTHON_BACKEND: Backend = { + spec: PYTHON_SPEC, + fileExtension: PYTHON_CONFIG.fileExtension, + printModule, + importPath, + support: supportModule, + errorsModule, + renderType: pyType, + comment: "#", + driver: driverFiles, + supportImport: (needs, usesEnv, builtins) => { + const names = builtins.slice(); + if (usesEnv && needs.env) names.push("Capabilities"); + return [{ from: "SUPPORT", names: names.sort() }]; + }, + defaultCapabilities: { + ref: { kind: "name", name: "DEFAULT_CAPABILITIES" }, + imports: [{ from: "SUPPORT", names: ["DEFAULT_CAPABILITIES"] }], + seamName: (publicName) => `${publicName}_with`, + }, + // Aggressive: CPython pays a full frame per call (no JIT to elide it), so a chain like + // `random_digit` → `random_below` → `next_u32`, called once per digit, is the largest measured + // cost in `generateCpf`/`generateCnpj` (`engine/docs/progress.md` §8). `randomBelow`'s own body + // (a bounded rejection-sampling loop) is the largest shape this needs to reach, at 6 statements. + inlineBudget: { maxStatements: 12, rounds: 3 }, +}; diff --git a/engine/src/targets/rust/index.ts b/engine/src/targets/rust/index.ts new file mode 100644 index 000000000..43cd99f0f --- /dev/null +++ b/engine/src/targets/rust/index.ts @@ -0,0 +1,2807 @@ +/** + * The Rust backend. + * + * Rust 2021, standard library only. `std` has no HTTP client and no regex engine, which is the + * falsification exercise `docs/targets/rust-sketch.md` predicted; both frictions are handled + * without a crate (a generated `Capabilities` trait for the first, a dedicated straight-line + * scanner per pattern for the second, with an allocation-free backtracking matcher as the fallback + * for a pattern shape a scanner cannot cover — see the "Regex" section below). + * + * Ownership. The shared Target AST carries no lifetimes, so this backend does not attempt the + * sketch's `&str` parameters: every heap value (`String`, `Vec`, a record) is owned wherever it + * is bound — parameters, locals and struct fields alike. A borrow appears only where Rust supplies + * one for free (a method call's `&self`, a `for` loop over `.iter()`) or where converting to owned + * would be pointless (a fresh value returned by a call). The one recurring cost is `.to_owned()` at + * points where a value has to be duplicated to satisfy the borrow checker; `docs/targets/rust.md` + * has the accounting. `.to_owned()` is used uniformly rather than `.clone()` because it is exactly + * as correct on a reference (`&str` → `String`) as on an owned value (`String` → `String`, via the + * blanket `Clone` impl), so the printer never has to know which one it is looking at. + * + * `Fail` is `Result`, one flat enum for the whole program rather than one type + * per utility (the sketch's plan): every fallible generated function shares the same error type, + * so a nested fallible call is exactly `let value = f(...)?;` — the idiomatic form, and simpler + * than the sketch expected because there is never a type mismatch for `?` to bridge. + */ + +import { computeBorrowableParams } from "../../analysis/borrows.ts"; +import type { Backend, DriverEntry, SupportNeeds } from "../../backend/generate.ts"; +import { ENGINE_VERSION } from "../../backend/generate.ts"; +import type { TargetSpec } from "../../backend/lower.ts"; +import type { Candidate } from "../../backend/select.ts"; +import { LoweringTable, argIsAscii } from "../../backend/select.ts"; +import type { TExpr, TFunc, TModule, TRecord, TStmt } from "../../backend/tast.ts"; +import type { CProgram } from "../../core/ir.ts"; +import type { CharRange, RegexNode } from "../../regex.ts"; +import { complement } from "../../regex.ts"; +import { BUILTIN_RECORDS, TRIM_CODE_POINTS } from "../../intrinsics/index.ts"; +import type { SemType } from "../../types.ts"; +import type { Value } from "../../values.ts"; + +export const RUST_CONFIG = { + baseline: "Rust 2021", + fileExtension: ".rs", + dependencies: [] as string[], + formatter: "rustfmt", + linters: ["cargo clippy"], +}; + +/* ------------------------------------------------------------------ * + * Naming + * ------------------------------------------------------------------ */ + +/** Rust keywords a snake_cased source identifier can collide with (`type`, from `HolidayType`). */ +const RUST_KEYWORDS = new Set([ + "as", "break", "const", "continue", "crate", "dyn", "else", "enum", "extern", "false", "fn", "for", + "if", "impl", "in", "let", "loop", "match", "mod", "move", "mut", "pub", "ref", "return", "self", + "Self", "static", "struct", "super", "trait", "true", "type", "unsafe", "use", "where", "while", + "async", "await", "dyn", "abstract", "become", "box", "do", "final", "macro", "override", "priv", + "typeof", "unsized", "virtual", "yield", "try", +]); + +/** Escapes a Rust keyword with the raw-identifier prefix, so `type` becomes a usable field name. */ +function escapeKeyword(name: string): string { + return RUST_KEYWORDS.has(name) ? `r#${name}` : name; +} + +/** camelCase or PascalCase to snake_case; the source language is always one of the two. */ +function snake(name: string): string { + const cleaned = name.replaceAll("$", "_").replace(/[-](.)/g, (_m, char: string) => char.toUpperCase()); + const snaked = cleaned + .replace(/([a-z0-9])([A-Z])/g, "$1_$2") + .replace(/([A-Z]+)([A-Z][a-z])/g, "$1_$2") + .toLowerCase(); + return escapeKeyword(snaked); +} + +function pascal(name: string): string { + const cleaned = name.replaceAll("$", "_").replace(/[-_](.)/g, (_m, char: string) => char.toUpperCase()); + return cleaned.charAt(0).toUpperCase() + cleaned.slice(1); +} + +/** The module's flat file identifier: also its `mod` name, so it has to be a valid identifier. */ +function rustModuleName(sourcePath: string): string { + return sourcePath.split("/").join("_").replaceAll("-", "_"); +} + +/* ------------------------------------------------------------------ * + * Types + * ------------------------------------------------------------------ */ + +export function rustType(type: SemType): string { + switch (type.kind) { + case "Bool": + return "bool"; + case "Int": + case "Decimal": + case "CivilDate": + case "Instant": + case "Duration": + return "i64"; + case "Float": + return "f64"; + case "String": + case "Enum": + return "String"; + case "List": + return `Vec<${rustType(type.elem)}>`; + case "Option": + return `Option<${rustType(type.inner)}>`; + case "Record": + // The one record the engine defines that is not a plain struct: the capability + // environment is threaded as a trait object, never constructed by generated code. + return type.name === "Capabilities" ? "&dyn Capabilities" : pascal(type.name); + case "Union": + return pascal(type.name); + case "Lambda": + return `Box ${rustType(type.ret)}>`; + case "Void": + return "()"; + case "Never": + return "()"; + default: { + const exhaustive: never = type; + return exhaustive; + } + } +} + +/** Whether a value of this type is `Copy`: never needs `.to_owned()`, and `.clone()` on it warns. */ +function isCopyType(type: SemType): boolean { + switch (type.kind) { + case "Bool": + case "Int": + case "Float": + case "Decimal": + case "CivilDate": + case "Instant": + case "Duration": + case "Void": + case "Never": + return true; + case "Option": + return isCopyType(type.inner); + default: + return false; + } +} + +/* ------------------------------------------------------------------ * + * String and identifier literals + * ------------------------------------------------------------------ */ + +/** Escapes a value as a Rust `&str` literal, ASCII only (Rust's escape syntax, not Go's/JS's). */ +function rustString(value: string): string { + let out = '"'; + for (const scalar of value) { + const point = scalar.codePointAt(0)!; + if (scalar === '"') out += '\\"'; + else if (scalar === "\\") out += "\\\\"; + else if (point === 0x0a) out += "\\n"; + else if (point === 0x0d) out += "\\r"; + else if (point === 0x09) out += "\\t"; + else if (point < 0x20 || point > 0x7e) out += `\\u{${point.toString(16)}}`; + else out += scalar; + } + return `${out}"`; +} + +/* ------------------------------------------------------------------ * + * Record field types, for deciding when a field value needs `.to_owned()`. + * + * Rebuilt at the start of every `printModule` call from that module's own `TRecord`s, and seeded + * once with the engine's own builtin records (`HttpRequest` and friends), which no module lists + * in `TModule.records` under its own name. + * ------------------------------------------------------------------ */ + +const recordFieldTypes = new Map>(); +for (const record of BUILTIN_RECORDS) { + recordFieldTypes.set( + record.name, + new Map(record.fields.map((field) => [field.name, field.optional ? { kind: "Option", inner: field.type } : field.type])), + ); +} + +function rebuildRecordFieldTypes(records: readonly TRecord[]): void { + for (const record of records) { + recordFieldTypes.set(record.name, new Map(record.fields.map((field) => [field.name, field.type]))); + } +} + +/** + * A hoisted constant table (`hoistConstantTables` in the shared lowerer) is declared + * `pub const NAME: &[T]`, Rust's own SCREAMING_SNAKE_CASE convention for a module-level constant; + * the reference the lowerer rewrote the literal into keeps the original lowercase name, so these + * two maps (rebuilt per module in `printModule`) are what let the printer reconcile the casing, + * and know a constant's element type when a call needs `.to_owned()` to turn `&[T]` into `Vec`. + */ +const hoistedConstantNames = new Set(); +const hoistedConstantTypes = new Map(); + +function printedName(name: string): string { + return hoistedConstantNames.has(name) ? name.toUpperCase() : name; +} + +/* ------------------------------------------------------------------ * + * Ownership: converting a value that may be a reference into one the target position owns. + * + * `print` never needs this by itself (see the module comment): a method call borrows its + * receiver, a binary operator borrows both sides, an intrinsic's own text decides its own + * borrowing. Only the handful of positions that build an owned aggregate — a record field, a list + * item, `Some(...)`, a `let`/`assign`/`return` value, a call argument — ever need it, and each of + * those calls `toOwned` with the type it expects there. + * ------------------------------------------------------------------ */ + +function toOwned(expr: TExpr, expectedType: SemType | undefined): string { + if (expectedType !== undefined && isCopyType(expectedType)) return print(expr); + switch (expr.kind) { + case "name": + // The capability environment is `&dyn Capabilities`, never owned by generated code. + if (expr.name === "env") return expr.name; + return `${printedName(expr.name)}.to_owned()`; + case "member": + // Reuses `print`'s own "member" case (rather than repeating its target rendering here) + // so a narrowed `Option` read goes through the same `.as_ref().unwrap()` rewrite either + // way — whether the read stands alone or, as here, feeds an owning position. + return `${print(expr)}.to_owned()`; + case "lit": + if (typeof expr.value === "string") return `${rustString(expr.value)}.to_string()`; + if (Array.isArray(expr.value)) return print(expr); + return print(expr); + case "ternary": + return `(if ${print(expr.test)} { ${toOwned(expr.then, expectedType)} } else { ${toOwned(expr.otherwise, expectedType)} })`; + case "some": { + const inner = expectedType !== undefined && expectedType.kind === "Option" ? expectedType.inner : undefined; + return `Some(${toOwned(expr.inner, inner)})`; + } + case "none": + return "None"; + default: + // call, method, raw, binary, unary, index, record, list, lambda, zero: every one of + // these already produces a fresh, owned value, so there is nothing to convert. + return print(expr); + } +} + +/** The type of a place expression, resolved as far as the record field table can take it. */ +function resolveType(expr: TExpr, scope: ReadonlyMap): SemType | undefined { + switch (expr.kind) { + case "name": + return scope.get(expr.name) ?? hoistedConstantTypes.get(expr.name); + case "member": { + const targetType = resolveType(expr.target, scope); + if (targetType === undefined || targetType.kind !== "Record") return undefined; + return recordFieldTypes.get(targetType.name)?.get(expr.name); + } + case "lit": + case "list": + return expr.type; + default: + return undefined; + } +} + +/** + * A call argument: owned when the callee's parameter is (every generated parameter is owned). + * When the type cannot be resolved — a candidate's own fragment reaching a nested call before any + * scope exists (see `currentScope`) — `toOwned` still converts a bare name or field read, just + * without being able to skip a `Copy` value's conversion as an optimization. That is never a + * correctness problem (`.to_owned()` compiles, and is a plain unmodified copy, for any `Clone` + * type, `Copy` ones included) and never a lint problem either: `clippy::clone_on_copy` matches the + * method name `clone`, not `to_owned`, which is exactly why `toOwned` reaches for the latter. + */ +function callArg(expr: TExpr, scope: ReadonlyMap): string { + return toOwned(expr, resolveType(expr, scope)); +} + +/* ------------------------------------------------------------------ * + * Regex: a dedicated matcher per pattern, compiled at engine build time. + * + * `std` has no regex engine (the sketch's second predicted friction). The engine knows every + * pattern a project uses before it generates a single line of Rust, so there is no reason to make + * the binary interpret a pattern tree at call time at all: each normalized pattern is compiled + * here into a straight-line matching function (`re_match_N`, below) that only ever tests the exact + * classes and counts that pattern names, in the order it names them. This is the fix the + * coordinator's benchmark asked for (Go's `re.test` recompiles from source on every call): there + * is no per-call compilation step to avoid here, because there is no run-time representation of + * the pattern left to interpret. + * + * A "chain" pattern — the shape every pattern in this project actually has — is a top-level + * sequence of plain character classes and repeated character classes, with no alternation and no + * repeated group. `chainElementsOf` recognizes this shape and additionally requires that any + * variable-length run (anything but an exact `{n}` or a bare class) have a class disjoint from + * whatever immediately follows it: that is the maximal-munch property that lets the generated + * scanner consume a run greedily, to the end of its own class, and never need to back off — the + * two adjacent classes can never disagree about where one run ends and the next begins. Every + * pattern `core/source` uses is exactly this: a fixed-count digit or alnum run, then a + * variable-count mask-character run disjoint from it, repeated. `renderChainScanner` turns the + * recognized element list directly into a `pub fn` built from two tiny generic helpers + * (`re_take_fixed`/`re_take_class`, in `GENERIC_SUPPORT`) that slice `&str` forward once, with no + * allocation anywhere. + * + * A pattern `chainElementsOf` refuses — alternation anywhere, a repeated group more complex than + * one class, or two adjacent variable-length runs whose classes could overlap — falls back to + * `re_test`/`ReNode` in `support.rs`: a backtracking matcher, structured exactly like `regex.ts`'s + * own reference `matchNode` (greedy, try one more repetition before giving up and continuing), but + * over `&str` byte slices instead of a collected `Vec`, so it never allocates either. No + * pattern in `core/source` takes this path today (see `LOWERING.md`'s `re.test` row and + * `docs/targets/rust.md` for the rule this section implements), but the accepted regex subset is + * not fully covered by the scanner, and soundness for whatever a project's patterns turn out to be + * matters more than never emitting the fallback. + * ------------------------------------------------------------------ */ + +/** One run in a chain pattern: `min === max` (and neither is `null`) means a fixed-count run. */ +type ChainElement = { + readonly ranges: readonly CharRange[]; + readonly negated: boolean; + readonly min: number; + readonly max: number | null; +}; + +function isFixed(element: ChainElement): boolean { + return element.max !== null && element.max === element.min; +} + +/** The positive code point set a class actually matches, negation resolved. */ +function effectiveRanges(ranges: readonly CharRange[], negated: boolean): CharRange[] { + return negated ? complement(ranges) : [...ranges]; +} + +/** Both range lists are already sorted and merged (every class is, from `regex.ts`'s parser). */ +function rangesDisjoint(a: readonly CharRange[], b: readonly CharRange[]): boolean { + let i = 0; + let j = 0; + while (i < a.length && j < b.length) { + const x = a[i]!; + const y = b[j]!; + if (x.hi < y.lo) i++; + else if (y.hi < x.lo) j++; + else return false; + } + return true; +} + +/** + * Recognizes the "chain" shape (see the section comment) and answers its elements left to right, + * or `undefined` when the pattern needs the fallback matcher instead. + */ +function chainElementsOf(node: RegexNode): ChainElement[] | undefined { + const items: readonly RegexNode[] = node.kind === "seq" ? node.items : [node]; + const elements: ChainElement[] = []; + for (const item of items) { + if (item.kind === "class") { + elements.push({ ranges: item.ranges, negated: item.negated, min: 1, max: 1 }); + } else if (item.kind === "repeat" && item.item.kind === "class") { + // A `{0}` member matches nothing and never advances; it is pathological enough (and + // absent from every real pattern) that falling the whole thing back is simpler than + // reasoning about it here. + if (item.min === 0 && item.max === 0) return undefined; + elements.push({ ranges: item.item.ranges, negated: item.item.negated, min: item.min, max: item.max }); + } else { + // Alternation, or a repeated group that is itself a sequence/alternation/repeat. + return undefined; + } + } + for (let i = 0; i < elements.length; i++) { + const element = elements[i]!; + if (isFixed(element)) continue; + if (i === elements.length - 1) continue; // nothing follows: consume the rest, no ambiguity + const next = elements[i + 1]!; + // Two adjacent variable-length runs would need to negotiate how much each one takes; the + // maximal-munch argument above only settles that question one run at a time. + if (!isFixed(next)) return undefined; + const ownSet = effectiveRanges(element.ranges, element.negated); + const nextSet = effectiveRanges(next.ranges, next.negated); + if (!rangesDisjoint(ownSet, nextSet)) return undefined; + } + return elements; +} + +/** `c: u32` inline range test, the same idiom `re.retain` and `manual_range_contains` both use. */ +function classPredicateExpr(ranges: readonly CharRange[], negated: boolean): string { + const test = ranges + .map((range) => (range.lo === range.hi ? `c == ${range.lo}` : `(${range.lo}..=${range.hi}).contains(&c)`)) + .join(" || "); + const body = test === "" ? "false" : test; + return negated ? `!(${body})` : body; +} + +function renderChainScanner(name: string, elements: readonly ChainElement[]): string { + const lines: string[] = [`pub fn ${name}(value: &str) -> bool {`]; + if (elements.length === 0) { + // The empty chain: a pattern that only ever matches the empty string. + lines.push("\tvalue.is_empty()", "}"); + return lines.join("\n"); + } + lines.push("\tlet rest = value;"); + for (const element of elements) { + const predicate = `|c: u32| ${classPredicateExpr(element.ranges, element.negated)}`; + if (isFixed(element)) { + lines.push(`\tlet Some(rest) = re_take_fixed(rest, ${element.min}, ${predicate}) else { return false; };`); + } else { + const max = element.max === null ? "usize::MAX" : String(element.max); + lines.push(`\tlet Some(rest) = re_take_class(rest, ${element.min}, ${max}, ${predicate}) else { return false; };`); + } + } + lines.push("\trest.is_empty()", "}"); + return lines.join("\n"); +} + +function renderReNode(node: RegexNode): string { + switch (node.kind) { + case "class": { + const ranges = node.ranges.map((range) => `(${range.lo}, ${range.hi})`).join(", "); + return `ReNode::Class(&[${ranges}], ${node.negated})`; + } + case "seq": + return `ReNode::Seq(&[${node.items.map(renderReNode).join(", ")}])`; + case "alt": + return `ReNode::Alt(&[${node.options.map(renderReNode).join(", ")}])`; + case "repeat": + return `ReNode::Repeat(&${renderReNode(node.item)}, ${node.min}, ${node.max === null ? "None" : `Some(${node.max})`})`; + default: { + const exhaustive: never = node; + return exhaustive; + } + } +} + +type RePattern = + | { readonly kind: "scanner"; readonly name: string; readonly rust: string } + | { readonly kind: "fallback"; readonly name: string; readonly rust: string }; + +const rePatterns = new Map(); +/** Whether any registered pattern took the fallback path — `supportModule` uses this to decide + * whether `RE_MATCHER_SUPPORT` (dead weight otherwise) belongs in the emitted `support.rs`. */ +let reFallbackUsed = false; + +/** Registers (or reuses) the matcher for one regex, and answers its Rust identifier and kind. */ +function registerPattern(node: RegexNode, source: string): RePattern { + const existing = rePatterns.get(source); + if (existing !== undefined) return existing; + const index = rePatterns.size; + const elements = chainElementsOf(node); + const pattern: RePattern = + elements !== undefined + ? { kind: "scanner", name: `re_match_${index}`, rust: renderChainScanner(`re_match_${index}`, elements) } + : { + kind: "fallback", + name: `RE_PATTERN_${index}`, + rust: `pub static RE_PATTERN_${index}: ReNode = ${renderReNode(node)};`, + }; + if (pattern.kind === "fallback") reFallbackUsed = true; + rePatterns.set(source, pattern); + return pattern; +} + +/* ------------------------------------------------------------------ * + * Capability table + * ------------------------------------------------------------------ */ + +const raw = (text: string): TExpr => ({ kind: "raw", text }); + +function binary(op: string): Candidate["emit"] { + return (args) => ({ kind: "binary", op, left: args[0]!, right: args[1]! }); +} + +/** + * An index or length as `usize`, the type every Rust indexing operation needs (the Core's own + * index type is `i64`, matching every other integer). A literal is printed bare rather than cast: + * Rust infers a bare integer literal's type from context (here, always `usize`, since indexing + * demands it), and `clippy::unnecessary_cast` (default warn) is exactly what flags `12 as usize` + * once that inference already gets there for free. + */ +function asUsize(expr: TExpr): string { + if (expr.kind === "lit" && typeof expr.value === "bigint") return expr.value.toString(); + return `${print(expr)} as usize`; +} + +/** + * A borrow of a value for a support helper's `&str`/`&[T]` parameter (see `GENERIC_SUPPORT`'s own + * comment on why every one of those borrows). A string literal is already `&'static str`, so + * borrowing it again is a redundant double reference — harmless to run, but exactly what + * `clippy::needless_borrow` (default warn) exists to catch — so this skips the `&` for one. + */ +function borrowed(expr: TExpr): string { + if (expr.kind === "lit" && typeof expr.value === "string") return print(expr); + // A candidate's own fragment (`str.trim`, say) ends in `.to_string()` because it is written to + // be self-sufficient wherever it lands — including an owning position, where that is exactly + // right. Immediately borrowing the String it just allocated is never right, though + // (`clippy::unnecessary_to_owned`, default warn, is what catches it), and every case this + // backend produces can just drop the conversion: what is left already borrows the original. + if (expr.kind === "raw" && expr.text.endsWith(".to_string()")) { + return expr.text.slice(0, -".to_string()".length); + } + // Likewise a fresh list literal: `&vec![1, 2, 3]` heap-allocates a `Vec` only to borrow it + // once: `&[1, 2, 3]` is the same `&[T]` argument, `clippy::useless_vec`'s (default warn) point. + if (expr.kind === "list") { + const elem = expr.type.kind === "List" ? expr.type.elem : undefined; + return `&[${expr.items.map((item) => toOwned(item, elem)).join(", ")}]`; + } + // A name already carrying its own reference — a parameter the borrow pre-pass found read-only + // (`expr.borrowed`, set at lowering time; see `docs/decisions/0010-*.md`) or a hoisted constant + // table (always `&[T]`, never owned) — needs no second `&`; that would be `&&str`/`&&[T]`, + // exactly what `clippy::needless_borrow` (default warn) exists to catch, same as the cases above. + if (expr.kind === "name" && (expr.borrowed === true || hoistedConstantNames.has(expr.name))) { + return printedName(expr.name); + } + return `&${print(expr)}`; +} + +/** A one-scalar string literal as a Rust `char` literal, or `undefined` for anything else. */ +function singleCharLiteral(expr: TExpr): string | undefined { + if (expr.kind !== "lit" || typeof expr.value !== "string") return undefined; + const scalars = [...expr.value]; + if (scalars.length !== 1) return undefined; + const scalar = scalars[0]!; + if (scalar === "\\") return "'\\\\'"; + if (scalar === "'") return "'\\''"; + if (scalar === "\n") return "'\\n'"; + if (scalar === "\t") return "'\\t'"; + if (scalar === "\r") return "'\\r'"; + return `'${scalar}'`; +} + +/** `Ordering`'s discriminants are exactly -1/0/1, so a cast is the whole comparison. */ +function orderingAsInt(a: TExpr, b: TExpr): TExpr { + return raw(`(${print(a)}.cmp(&${print(b)}) as i64)`); +} + +const cheap = { alloc: "none", time: "constant" } as const; +const linear = { alloc: "none", time: "linear" } as const; +const allocating = { alloc: "one", time: "linear" } as const; +const scalarPass = { alloc: "many", time: "linear" } as const; + +export const RUST_CANDIDATES: readonly Candidate[] = [ + ...["add:+", "sub:-", "mul:*"].map((entry) => { + const [op, symbol] = entry.split(":") as [string, string]; + return { op: `int.${op}`, impl: "native" as const, cost: cheap, emit: binary(symbol) }; + }), + { + op: "int.div", + impl: "native", + because: "Rust's `/` truncates toward zero, which is the Core's rule", + cost: cheap, + emit: binary("/"), + }, + { + op: "int.mod", + impl: "native", + because: "Rust's `%` takes the sign of the dividend, which is the Core's rule", + cost: cheap, + emit: binary("%"), + }, + ...["add:+", "sub:-", "mul:*", "div:/"].map((entry) => { + const [op, symbol] = entry.split(":") as [string, string]; + return { op: `float.${op}`, impl: "native" as const, cost: cheap, emit: binary(symbol) }; + }), + { op: "int.neg", impl: "native", cost: cheap, emit: (args) => raw(`-${print(args[0]!)}`) }, + { op: "float.neg", impl: "native", cost: cheap, emit: (args) => raw(`-${print(args[0]!)}`) }, + { op: "int.abs", impl: "native", cost: cheap, emit: (args) => raw(`${print(args[0]!)}.abs()`) }, + { + op: "int.min", + impl: "native", + cost: cheap, + // `int.max(x, lo)` then `int.min(…, hi)` is how the source spells a clamp (there is no + // clamp intrinsic); printed as two separate calls that is `x.max(lo).min(hi)`, exactly the + // shape `clippy::manual_clamp` (default warn) asks to become `.clamp(lo, hi)`. Recognizing + // it here, rather than adding a `clamp` intrinsic, keeps that merge a printer concern. + emit: (args) => { + const inner = args[0]!; + if (inner.kind === "method" && inner.name === "max" && inner.args.length === 1) { + return raw(`${print(inner.target)}.clamp(${print(inner.args[0]!)}, ${print(args[1]!)})`); + } + return { kind: "method", target: inner, name: "min", args: [args[1]!] }; + }, + }, + { + op: "int.max", + impl: "native", + cost: cheap, + emit: (args) => { + const inner = args[0]!; + if (inner.kind === "method" && inner.name === "min" && inner.args.length === 1) { + return raw(`${print(inner.target)}.clamp(${print(args[1]!)}, ${print(inner.args[0]!)})`); + } + return { kind: "method", target: inner, name: "max", args: [args[1]!] }; + }, + }, + ...["lt:<", "le:<=", "gt:>", "ge:>="].flatMap((entry) => { + const [op, symbol] = entry.split(":") as [string, string]; + return [ + { op: `int.${op}`, impl: "native" as const, cost: cheap, emit: binary(symbol) }, + { op: `float.${op}`, impl: "native" as const, cost: cheap, emit: binary(symbol) }, + ]; + }), + { op: "float.fromInt", impl: "native", cost: cheap, emit: (args) => raw(`(${print(args[0]!)} as f64)`) }, + { op: "core.eq", impl: "native", cost: cheap, emit: binary("==") }, + + { + op: "opt.isNone", + impl: "native", + cost: cheap, + emit: (args) => ({ kind: "method", target: args[0]!, name: "is_none", args: [] }), + }, + { + op: "opt.unwrap", + impl: "native", + cost: cheap, + emit: (args) => ({ kind: "method", target: args[0]!, name: "unwrap", args: [] }), + }, + { op: "opt.some", impl: "native", cost: cheap, emit: (args) => ({ kind: "some", inner: args[0]! }) }, + { + op: "opt.orElse", + impl: "native", + cost: cheap, + // `unwrap_or`'s fallback has to match the `Option`'s own inner type exactly (a bare string + // literal is `&str`, not the `String` an `Option` needs), which is the one place a + // candidate's own argument genuinely needs `toOwned` rather than a plain `print`. + emit: (args, types) => { + const inner = types[0]?.kind === "Option" ? types[0].inner : undefined; + return raw(`${print(args[0]!)}.unwrap_or(${toOwned(args[1]!, inner)})`); + }, + }, + + { + op: "str.len", + impl: "native", + requires: argIsAscii(0), + because: "`str::len` counts bytes, which equals the scalar count only for ASCII", + cost: cheap, + emit: (args) => raw(`(${print(args[0]!)}.len() as i64)`), + }, + { + op: "str.len", + impl: "native", + because: "`chars().count()` walks scalars without allocating, unlike Go's `[]rune` conversion", + cost: linear, + emit: (args) => raw(`(${print(args[0]!)}.chars().count() as i64)`), + }, + { + op: "str.concat", + impl: "library", + because: "`concat2` borrows both operands, unlike `+`, which would consume the left one", + cost: allocating, + // Not `format!("{}{}", a, b)`: the Core builds a multi-piece join as nested binary + // concatenations, so a template literal with several interpolations nests one `format!` + // call inside another's arguments, which `clippy::format_in_format_args` (default warn) + // catches every time. A plain function call nests without that concern. + emit: (args) => raw(`crate::support::concat2(${borrowed(args[0]!)}, ${borrowed(args[1]!)})`), + }, + { + // `backend/lower.ts`'s `operation` flattens a `+` chain of three or more pieces into this + // before lowering (see its own comment); every intermediate `str.concat` in that chain would + // otherwise have reallocated and copied everything to its left, once per additional piece — + // `format_currency`'s assembly of `prefix`/`sign`/`body` and each `concat2` in + // `random_cpf_base`'s nine-digit chain were exactly that (`engine/docs/progress.md` §8). One + // buffer, sized once from every piece's own length, replaces the whole chain. + op: "str.concatAll", + impl: "native", + because: "one buffer sized once, not a chain of reallocate-and-copy", + cost: allocating, + emit: (args) => { + // A piece's length is needed once to size the buffer and once more to push it. `.len()` + // auto-derefs, so the bare (unborrowed) form works for sizing regardless of whether a piece + // ends up owned or borrowed — the borrow only matters for `push_str`, which needs a `&str`. + // A name or a literal is cheap to print twice this way (it is just a reference, or already + // a constant); anything else — a call, most often — is bound to a local first, so what it + // computes runs once, not twice. + const bindings: string[] = []; + const pieces = args.map((arg, index) => { + if (arg.kind === "lit") return { bare: print(arg), borrow: print(arg), char: singleCharLiteral(arg) }; + if (arg.kind === "name") return { bare: print(arg), borrow: borrowed(arg), char: undefined }; + const local = `__piece${index}`; + bindings.push(`let ${local} = ${print(arg)};`); + return { bare: local, borrow: `&${local}`, char: undefined }; + }); + const capacity = pieces.map((piece) => `${piece.bare}.len()`).join(" + "); + // `clippy::single_char_add_str` (default warn) wants `push('x')` over `push_str("x")` for a + // one-character literal — the one case a piece's own length is already known, too. + const pushes = pieces + .map((piece) => (piece.char !== undefined ? `__buf.push(${piece.char});` : `__buf.push_str(${piece.borrow});`)) + .join(" "); + return raw(`{ ${bindings.join(" ")} let mut __buf = String::with_capacity(${capacity}); ${pushes} __buf }`); + }, + }, + { + op: "str.codeAt", + impl: "native", + requires: argIsAscii(0), + because: "indexing a byte string yields a byte", + cost: cheap, + emit: (args) => raw(`(${print(args[0]!)}.as_bytes()[${asUsize(args[1]!)}] as i64)`), + }, + { + op: "str.charAt", + impl: "native", + requires: argIsAscii(0), + cost: allocating, + emit: (args) => raw(`(${print(args[0]!)}.as_bytes()[${asUsize(args[1]!)}] as char).to_string()`), + }, + { + op: "str.codeAtOpt", + impl: "library", + requires: argIsAscii(0), + cost: cheap, + emit: (args) => raw(`crate::support::code_at(${borrowed(args[0]!)}, ${print(args[1]!)})`), + }, + { + op: "str.charAtOpt", + impl: "library", + requires: argIsAscii(0), + cost: allocating, + emit: (args) => raw(`crate::support::char_at(${borrowed(args[0]!)}, ${print(args[1]!)})`), + }, + { + op: "str.slice", + impl: "native", + requires: argIsAscii(0), + because: "slicing an ASCII string cuts at byte boundaries, which are scalar boundaries too", + cost: allocating, + emit: (args) => + raw(`${print(args[0]!)}[${asUsize(args[1]!)}..${asUsize(args[2]!)}].to_string()`), + }, + { + op: "str.indexOf", + impl: "native", + requires: argIsAscii(0), + because: "`str::find` answers a byte offset", + cost: linear, + emit: (args) => + raw(`${print(args[0]!)}.find(${print(args[1]!)}).map_or(-1, |byte| byte as i64)`), + }, + { + op: "str.contains", + impl: "native", + cost: linear, + emit: (args) => ({ kind: "method", target: args[0]!, name: "contains", args: [args[1]!] }), + }, + { + op: "str.startsWith", + impl: "native", + cost: linear, + emit: (args) => ({ kind: "method", target: args[0]!, name: "starts_with", args: [args[1]!] }), + }, + { + op: "str.endsWith", + impl: "native", + cost: linear, + emit: (args) => ({ kind: "method", target: args[0]!, name: "ends_with", args: [args[1]!] }), + }, + { + op: "str.repeat", + impl: "native", + cost: allocating, + emit: (args) => raw(`${print(args[0]!)}.repeat(${asUsize(args[1]!)})`), + }, + { + op: "str.padStart", + impl: "native", + requires: (args) => argIsAscii(0)(args) && argIsAscii(2)(args), + because: + "when both the value and the pad string are proven ASCII, a scalar count is a byte count, " + + "so the length check and the padding loop need no `Vec` at all -- `pad_start` below " + + "builds one just to learn `value.len()` and to hand `extend` something to iterate, which " + + "was measured at roughly half of `format_currency`'s own cost (`pad_start`'s call on the " + + "whole-part digits) for a value this short", + cost: allocating, + emit: (args) => + raw( + `crate::support::pad_start_ascii(${borrowed(args[0]!)}, ${print(args[1]!)}, ${borrowed(args[2]!)})`, + ), + }, + { + op: "str.padStart", + impl: "library", + cost: allocating, + emit: (args) => + raw(`crate::support::pad_start(${borrowed(args[0]!)}, ${print(args[1]!)}, ${borrowed(args[2]!)})`), + }, + { + op: "str.trim", + impl: "native", + because: "`trim_matches` takes the cut set explicitly, so the 25 code points are exact", + cost: allocating, + emit: (args) => + raw( + `${print(args[0]!)}.trim_matches(|c: char| matches!(c as u32, ${TRIM_CODE_POINTS.join(" | ")})).to_string()`, + ), + }, + { + op: "str.asciiUpper", + impl: "native", + requires: argIsAscii(0), + because: "`to_ascii_uppercase` is only ASCII-equivalent on ASCII input", + cost: allocating, + emit: (args) => ({ kind: "method", target: args[0]!, name: "to_ascii_uppercase", args: [] }), + }, + { + op: "str.asciiLower", + impl: "native", + requires: argIsAscii(0), + cost: allocating, + emit: (args) => ({ kind: "method", target: args[0]!, name: "to_ascii_lowercase", args: [] }), + }, + { + op: "str.asciiUpper", + impl: "native", + because: "mapping only a-z, leaving every other scalar alone, is the Core's rule for any input", + cost: allocating, + emit: (args) => + raw( + `${print(args[0]!)}.chars().map(|c| if c.is_ascii_lowercase() { c.to_ascii_uppercase() } else { c }).collect::()`, + ), + }, + { + op: "str.asciiLower", + impl: "native", + because: "mapping only A-Z, leaving every other scalar alone, is the Core's rule for any input", + cost: allocating, + emit: (args) => + raw( + `${print(args[0]!)}.chars().map(|c| if c.is_ascii_uppercase() { c.to_ascii_lowercase() } else { c }).collect::()`, + ), + }, + { + op: "str.compare", + impl: "native", + because: "`Ord` on `str` compares UTF-8 bytes, which is code point order — like Go, unlike JavaScript", + cost: linear, + emit: (args) => orderingAsInt(args[0]!, args[1]!), + }, + { + op: "str.codePoints", + impl: "native", + requires: argIsAscii(0), + because: + "an ASCII byte is already its own code point, so `.bytes()` needs no UTF-8 decode at all, " + + "unlike `code_points`' `.chars()` below", + cost: allocating, + emit: (args) => raw(`${print(args[0]!)}.bytes().map(|b| b as i64).collect::>()`), + }, + { + op: "str.codePoints", + impl: "library", + cost: allocating, + emit: (args) => raw(`crate::support::code_points(${borrowed(args[0]!)})`), + }, + { + op: "str.fromCodePoints", + impl: "native", + requires: (args) => { + const elem = args[0]; + return elem !== undefined && elem.kind === "List" && elem.elem.kind === "Int" && elem.elem.lo >= 0 && elem.elem.hi <= 127; + }, + because: + "every code point this project ever builds this way is proven ASCII (`group_thousands`'s " + + "`out`), so its value is its whole UTF-8 encoding -- pushed straight into the `String` this " + + "builds, one byte-as-char per point, instead of `from_code_points`'s `char::from_u32` round " + + "trip below, which decodes a full scalar this project never produces", + cost: allocating, + emit: (args) => + raw( + `{ let __pts = ${borrowed(args[0]!)}; let mut __out = String::with_capacity(__pts.len()); ` + + `for &__p in __pts { __out.push(__p as u8 as char); } __out }`, + ), + }, + { + op: "str.fromCodePoints", + impl: "library", + cost: allocating, + emit: (args) => raw(`crate::support::from_code_points(${borrowed(args[0]!)})`), + }, + { + op: "str.asAscii", + impl: "library", + cost: linear, + emit: (args) => raw(`crate::support::as_ascii(${borrowed(args[0]!)})`), + }, + { + op: "str.asDigits", + impl: "library", + cost: linear, + emit: (args) => raw(`crate::support::as_digits(${borrowed(args[0]!)})`), + }, + { + op: "str.split", + impl: "native", + cost: allocating, + emit: (args) => + raw(`${print(args[0]!)}.split(${print(args[1]!)}).map(str::to_string).collect::>()`), + }, + { + op: "str.join", + impl: "native", + cost: allocating, + emit: (args) => raw(`${print(args[0]!)}.join(${print(args[1]!)})`), + }, + { + op: "str.fromInt", + impl: "native", + requires: (args) => { + const arg = args[0]; + return arg !== undefined && arg.kind === "Int" && arg.lo >= 0n && arg.hi <= 9n; + }, + because: + "a value proven to be a single decimal digit (`randomDigit`'s own call, after " + + "specialization narrows `randomBelow`'s result at this call site) needs no general-purpose " + + "integer formatter -- one ASCII byte pushed directly is the whole job, where `i64::to_string` " + + "below computes a digit count, allocates a buffer sized for it, and writes back to front " + + "even for a single digit. A `call` to a small support function, not a `raw` fragment that " + + "prints its argument as text immediately: `hoistConstantTables` (`backend/lower.ts`) walks " + + "the *structured* Target AST for a list literal to lift into a module-level constant, and " + + "only runs after every candidate's own `emit`, so a `raw` fragment that has already flattened " + + "an argument's call tree into a source string -- which this candidate's argument sometimes " + + "is, e.g. `cnpj_check_digit`'s own weight-table argument in `generate_cnpj` -- would hide a " + + "list literal nested inside it from that pass, the same way it stayed hidden before this " + + "whole node was a `call` in disguise. `str.fromInt`'s own general candidate below keeps that " + + "same discipline (`method`, not `raw`), for the same reason.", + cost: allocating, + emit: (args) => ({ + kind: "call", + callee: { kind: "raw", text: "crate::support::digit_char" }, + args: [args[0]!], + }), + }, + { + op: "str.fromInt", + impl: "native", + cost: allocating, + emit: (args) => ({ kind: "method", target: args[0]!, name: "to_string", args: [] }), + }, + { + op: "str.parseInt", + impl: "library", + cost: linear, + emit: (args) => raw(`crate::support::parse_digits(${borrowed(args[0]!)})`), + }, + + { + op: "seq.at", + impl: "library", + cost: cheap, + emit: (args) => raw(`crate::support::at(${borrowed(args[0]!)}, ${print(args[1]!)})`), + }, + { + op: "seq.get", + impl: "library", + cost: cheap, + emit: (args) => raw(`crate::support::get_at(${borrowed(args[0]!)}, ${print(args[1]!)})`), + }, + { + op: "seq.len", + impl: "native", + cost: cheap, + emit: (args) => raw(`(${print(args[0]!)}.len() as i64)`), + }, + { + op: "seq.push", + impl: "native", + cost: cheap, + // Unlike every other method this table emits, `Vec::push` consumes its argument rather than + // borrowing it, so — alone among them — it needs `toOwned` rather than a plain `print`, the + // same reason `opt.orElse`'s fallback does above. + emit: (args) => raw(`${print(args[0]!)}.push(${toOwned(args[1]!, undefined)})`), + }, + { + op: "seq.sum", + impl: "native", + cost: linear, + emit: (args) => raw(`${print(args[0]!)}.iter().sum::()`), + }, + { + op: "seq.contains", + impl: "native", + cost: linear, + emit: (args) => raw(`${print(args[0]!)}.contains(${borrowed(args[1]!)})`), + }, + { + op: "seq.indexOf", + impl: "native", + cost: linear, + emit: (args) => + raw(`${print(args[0]!)}.iter().position(|item| item == ${borrowed(args[1]!)}).map_or(-1, |i| i as i64)`), + }, + { + op: "seq.concat", + impl: "native", + cost: allocating, + emit: (args) => + raw(`${print(args[0]!)}.iter().chain(${print(args[1]!)}.iter()).cloned().collect::>()`), + }, + { + op: "seq.slice", + impl: "native", + cost: allocating, + emit: (args) => + raw(`${print(args[0]!)}[${asUsize(args[1]!)}..${asUsize(args[2]!)}].to_vec()`), + }, + { + op: "seq.reverse", + impl: "native", + cost: allocating, + emit: (args) => raw(`${print(args[0]!)}.iter().rev().cloned().collect::>()`), + }, + { + op: "seq.sortStable", + impl: "native", + because: "`slice::sort` is a stable sort in Rust's own std, unlike Go's `sort.Slice`", + cost: { alloc: "one", time: "nlogn" }, + emit: (args) => raw(`crate::support::sorted_stable(${borrowed(args[0]!)})`), + }, + { + op: "seq.sortStableBy", + impl: "native", + cost: { alloc: "one", time: "nlogn" }, + emit: (args) => raw(`crate::support::sorted_stable_by(${borrowed(args[0]!)}, ${print(args[1]!)})`), + }, + { + op: "seq.map", + impl: "library", + cost: allocating, + emit: (args) => raw(`crate::support::mapped(${borrowed(args[0]!)}, ${print(args[1]!)})`), + }, + { + op: "seq.filter", + impl: "library", + cost: allocating, + emit: (args) => raw(`crate::support::filtered(${borrowed(args[0]!)}, ${print(args[1]!)})`), + }, + { + op: "seq.any", + impl: "native", + cost: linear, + emit: (args) => raw(`${print(args[0]!)}.iter().any(${print(args[1]!)})`), + }, + { + op: "seq.all", + impl: "native", + cost: linear, + emit: (args) => raw(`${print(args[0]!)}.iter().all(${print(args[1]!)})`), + }, + { + op: "seq.find", + impl: "native", + cost: linear, + emit: (args) => raw(`${print(args[0]!)}.iter().find(${print(args[1]!)}).cloned()`), + }, + + { op: "dec.fromScaled", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { + op: "dec.fromInt", + impl: "native", + cost: cheap, + emit: (args, types) => raw(`(${print(args[0]!)} * ${10 ** scaleOf(types[1])})`), + }, + { op: "dec.add", impl: "native", cost: cheap, emit: binary("+") }, + { op: "dec.sub", impl: "native", cost: cheap, emit: binary("-") }, + { op: "dec.mul", impl: "native", cost: cheap, emit: binary("*") }, + { op: "dec.compare", impl: "native", cost: cheap, emit: (args) => orderingAsInt(args[0]!, args[1]!) }, + { op: "dec.isNegative", impl: "native", cost: cheap, emit: (args) => raw(`(${print(args[0]!)} < 0)`) }, + { op: "dec.abs", impl: "native", cost: cheap, emit: (args) => raw(`${print(args[0]!)}.abs()`) }, + { op: "dec.unscaled", impl: "native", cost: cheap, emit: (args) => args[0]! }, + + { + op: "date.clampEpochDays", + impl: "native", + cost: cheap, + emit: (args) => raw(`${print(args[0]!)}.clamp(-719162, 2932896)`), + }, + { op: "date.toEpochDays", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { + op: "date.fromEpochDays", + impl: "library", + cost: cheap, + emit: (args) => raw(`crate::support::date_from_epoch_days(${print(args[0]!)})`), + }, + { + op: "date.fromYmd", + impl: "portable", + cost: linear, + sourceFn: "std/date::ymdToDays", + emit: (args, _types, ctx) => raw(`${portableCall("std/date::ymdToDays", ctx)}(${args.map(print).join(", ")})`), + }, + { + op: "date.year", + impl: "portable", + cost: cheap, + sourceFn: "std/date::yearFromDays", + emit: (args, _types, ctx) => raw(`${portableCall("std/date::yearFromDays", ctx)}(${print(args[0]!)})`), + }, + { + op: "date.month", + impl: "portable", + cost: cheap, + sourceFn: "std/date::monthFromDays", + emit: (args, _types, ctx) => raw(`${portableCall("std/date::monthFromDays", ctx)}(${print(args[0]!)})`), + }, + { + op: "date.day", + impl: "portable", + cost: cheap, + sourceFn: "std/date::dayFromDays", + emit: (args, _types, ctx) => raw(`${portableCall("std/date::dayFromDays", ctx)}(${print(args[0]!)})`), + }, + { + op: "date.addDays", + impl: "library", + cost: cheap, + emit: (args) => raw(`crate::support::date_from_epoch_days(${print(args[0]!)} + ${print(args[1]!)})`), + }, + { op: "date.diffDays", impl: "native", cost: cheap, emit: binary("-") }, + { op: "date.compare", impl: "native", cost: cheap, emit: (args) => orderingAsInt(args[0]!, args[1]!) }, + { + op: "date.dayOfWeek", + impl: "native", + cost: cheap, + emit: (args) => raw(`(((${print(args[0]!)} + 3).rem_euclid(7)) + 1)`), + }, + { + op: "date.isLeapYear", + impl: "native", + cost: cheap, + emit: (args) => + raw(`((${print(args[0]!)} % 4 == 0 && ${print(args[0]!)} % 100 != 0) || ${print(args[0]!)} % 400 == 0)`), + }, + + { + op: "re.retain", + impl: "native", + because: + "a byte-wise pass, not `.chars()`, when every retained range is ASCII (every class this " + + "project uses is: digits, letters) -- `.chars()` decodes the whole input as UTF-8 scalars " + + "before the filter ever runs, which was measured as the largest cost in `isValidCpf` once " + + "the regex engine itself stopped being one (`engine/docs/progress.md` §8); a byte never " + + "needs decoding to be range-tested, and a multi-byte scalar's bytes are all >= 0x80, so " + + "every one of them fails an ASCII range test on its own and is dropped exactly as it would " + + "be by testing the decoded scalar -- a non-ASCII input keeps working, just without ever " + + "paying to decode it. The ASCII branch writes straight into the `String` it returns with a " + + "plain `for` loop over `.bytes()`, rather than `.bytes().filter(...).collect::>()` " + + "followed by `String::from_utf8(..).unwrap()`: the iterator-adaptor chain and the `Vec` " + + "it builds only to hand to a UTF-8 validator that then has to re-walk it are both pure " + + "overhead here, since every byte the loop pushes is already a range-tested ASCII byte and " + + "therefore already valid UTF-8 on its own -- measured at roughly half the cost of the " + + "iterator-chain form on an 11-byte input (`is_valid_cpf`'s own `keep_digits` call), with no " + + "`unsafe` needed to get there.", + cost: allocating, + emit: (args, _types, ctx) => { + const ranges = ctx.regex === undefined || ctx.regex.node.kind !== "class" ? [] : ctx.regex.node.ranges; + // `(lo..=hi).contains(&c)`, not `c >= lo && c <= hi`: the same range the class already + // is, and what `clippy::manual_range_contains` (default warn) asks the latter to become. + const rangeTest = (name: string): string => + ranges.length === 0 + ? "false" + : ranges + .map((range) => (range.lo === range.hi ? `${name} == ${range.lo}` : `(${range.lo}..=${range.hi}).contains(&${name})`)) + .join(" || "); + if (ranges.every((range) => range.hi <= 127)) { + // `.len()` (to size the buffer) and `.bytes()` (to scan it) both need the source value, + // so it is bound to a local first when printing it twice would evaluate it twice -- the + // same rule `str.concatAll` above uses. A name or a literal is cheap to print twice (it + // is just a reference, or already a constant); anything else, most often a call, is + // bound once. + const arg = args[0]!; + const cheap = arg.kind === "lit" || arg.kind === "name"; + const source = cheap ? print(arg) : "__retain_src"; + const binding = cheap ? "" : `let ${source} = ${print(arg)}; `; + return raw( + `{ ${binding}let mut __out = String::with_capacity(${source}.len()); ` + + `for __b in ${source}.bytes() { if ${rangeTest("__b")} { __out.push(__b as char); } } __out }`, + ); + } + return raw( + `${print(args[0]!)}.chars().filter(|&ch| { let c = ch as u32; ${rangeTest("c")} }).collect::()`, + ); + }, + }, + { + op: "re.test", + // `std` has no regex engine, so this reaches into this backend's own support file — the + // same reason Go classifies its `padStart` (its own generated helper) as "library" rather + // than "native" — not into `std` itself, which is what "native" means throughout this table. + impl: "library", + because: + "a dedicated straight-line scanner (no allocation, one pass) when the pattern is a chain " + + "of character-class runs with no alternation and no two adjacent variable-length runs " + + "that could overlap; otherwise a backtracking matcher over a static pattern tree, also " + + "allocation-free — see engine/src/targets/rust/index.ts's \"Regex\" section for the rule", + cost: linear, + emit: (args, _types, ctx) => { + const source = ctx.regex?.source ?? ""; + const pattern = ctx.regex === undefined ? undefined : registerPattern(ctx.regex.node, source); + if (pattern === undefined) return raw("false"); + return pattern.kind === "scanner" + ? raw(`crate::support::${pattern.name}(${borrowed(args[0]!)})`) + : raw(`crate::support::re_test(&crate::support::${pattern.name}, ${borrowed(args[0]!)})`); + }, + }, + + { + op: "http.request", + impl: "native", + cost: { alloc: "many", time: "linear" }, + emit: (args, _types, ctx) => raw(`${print(ctx.env())}.request(${print(args[0]!)})`), + }, + { + op: "clock.now", + impl: "native", + cost: cheap, + emit: (_args, _types, ctx) => raw(`${print(ctx.env())}.now()`), + }, + { + op: "clock.sleep", + impl: "native", + cost: cheap, + emit: (args, _types, ctx) => raw(`${print(ctx.env())}.sleep(${print(args[0]!)})`), + }, + { op: "clock.millis", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { op: "clock.durationMillis", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { + op: "clock.elapsed", + impl: "native", + cost: cheap, + emit: (args) => raw(`(${print(args[1]!)} - ${print(args[0]!)}).max(0)`), + }, + { + op: "random.nextU32", + impl: "native", + cost: cheap, + emit: (_args, _types, ctx) => raw(`${print(ctx.env())}.next_u32()`), + }, + { + op: "task.race", + impl: "library", + because: "`std::thread::scope` plus an `mpsc` channel: one thread per task, first `Some` wins", + cost: { alloc: "many", time: "linear" }, + emit: (args) => { + const list = args[0]!; + if (list.kind !== "list") return raw(`crate::support::race_first_some(vec![${print(list)}])`); + const boxed = list.items + // Not `move`: a task closure only ever needs to read what it captures (an argument, the + // environment), and letting it borrow instead is what lets the *same* local be handed to + // every task — each capture is a shared reference, and any number of those can coexist. + .map((item) => `Box::new(|| ${printClosureBody(item)}) as Box _ + Send>`) + .join(", "); + return raw(`crate::support::race_first_some(vec![${boxed}])`); + }, + }, +]; + +/** A portable lowering's target sits in its own module, so the call needs a `crate::` path. */ +function portableCall(qualified: string, ctx: Parameters[2]): string { + const [modulePath] = qualified.split("::"); + return `crate::${rustModuleName(modulePath!)}::${ctx.nameOf(qualified)}`; +} + +/** The body of a lambda passed directly to `task.race`, as a closure body (`() -> Option`). */ +function printClosureBody(expr: TExpr): string { + if (expr.kind !== "lambda") return `{ ${print(expr)}() }`; + const body = withFnCtx({ ret: expr.ret, fails: [] }, () => printBody(expr.body, new Map())); + return `{\n${body}\n}`; +} + +function scaleOf(type: SemType | undefined): number { + return type !== undefined && type.kind === "Int" ? Number(type.lo) : 0; +} + +export const RUST_SPEC: TargetSpec = { + name: "rust", + table: new LoweringTable(RUST_CANDIDATES), + naming: { + func: (name) => snake(name), + value: (name) => snake(name), + field: (name) => snake(name), + type: (name) => pascal(name), + module: (path) => `src/${rustModuleName(path)}.rs`, + }, + loopCombinators: new Set(["seq.fold", "seq.map", "seq.filter"]), + statementTernary: false, + errorsAsValues: true, + asyncColouring: false, + envType: { kind: "Record", name: "Capabilities" }, +}; + +/* ------------------------------------------------------------------ * + * Printer + * + * `print` is scope-free by design (see the module comment): it is called both from inside a + * candidate's `emit` (lowering time, before any function's local scope exists) and from the + * statement printer below (print time). The handful of positions that need to know whether a + * value is owned or borrowed — record fields, list items, `return`, `let`, `assign`, call + * arguments — go through `toOwned`/`callArg` instead, which take the expected type explicitly. + * ------------------------------------------------------------------ */ + +export function print(expr: TExpr): string { + switch (expr.kind) { + case "lit": + return literal(expr.value, expr.type); + case "name": + return printedName(expr.name); + case "raw": + return expr.text; + case "call": { + // Cross-module calls are not qualified at the call site (unlike a portable lowering's + // or a support helper's, which this file writes by hand): a candidate's own `emit` can + // call `print` on an argument that itself contains an ordinary Core function call, + // which happens at lowering time, before any module's cross-module import data exists + // to qualify it with. `crate::*` (see `printModule`) is what makes the bare name resolve + // regardless of when the text was produced. + // + // `borrowedArgs[i]` (set at lowering time from the whole-program borrow map — see + // `docs/decisions/0010-*.md`) says the callee's own parameter there was found read-only; + // `borrowed()` is the same helper every support-function call already borrows its + // arguments with, reused here rather than duplicated. + const rendered = expr.args.map((arg, index) => + expr.borrowedArgs?.[index] === true ? borrowed(arg) : callArg(arg, currentScope), + ); + return `${print(expr.callee)}(${rendered.join(", ")})`; + } + case "method": + return `${print(expr.target)}.${snake(expr.name)}(${expr.args.map(print).join(", ")})`; + case "member": { + // A field read after a narrowed `Option` check (`if x === undefined { return … }`, then + // `x.field`) is exactly this shape: the Core never re-types the local narrower, so the + // `opt.unwrap` candidate still runs, once per read (see `docs/decisions/0009-*.md`). A + // plain `.unwrap()` would move the option out on the first read and leave the second one + // looking at a moved value, which is the one place `Option::unwrap`'s taking `self` by + // value — unlike a Go pointer dereference, which repeats for free — actually bites; going + // through `.as_ref()` borrows instead, and reads any number of times for the same reason + // a reference does everywhere else in this backend. + if (expr.target.kind === "method" && expr.target.name === "unwrap" && expr.target.args.length === 0) { + return `${print(expr.target.target)}.as_ref().unwrap().${expr.name}`; + } + return `${print(expr.target)}.${expr.name}`; + } + case "index": + return `${print(expr.target)}[${asUsize(expr.index)}]`; + case "binary": { + // `x == ""` / `x != ""` as `.is_empty()`: the Core has no dedicated emptiness check (an + // author writes `value.length === 0` or `value === ""`, both `core.eq`), so the rewrite + // belongs here rather than in a candidate. `clippy::comparison_to_empty` (default warn). + if (expr.op === "==" || expr.op === "!==" || expr.op === "!=") { + const empty = emptyStringCompare(expr); + if (empty !== undefined) return empty; + } + // `x >= lo && x <= hi` as `(lo..=hi).contains(&x)`, and the `||`-negated shape as the + // `!`-prefixed form: two source comparisons the Core has no range primitive to express + // directly (`docs/semantics.md` admits `<`/`<=`/`>`/`>=`, not a range type), so a bounds + // check is always written as the pair `clippy::manual_range_contains` (default warn) + // already knows the idiom for. + if (expr.op === "&&" || expr.op === "||") { + const range = rangeContains(expr); + if (range !== undefined) return range; + } + return `(${print(expr.left)} ${rustOperator(expr.op)} ${print(expr.right)})`; + } + case "unary": { + // `is_none()` negated reads as `is_some()`, and `!(a == b)` as `a != b` — the same + // rewrite Go's own printer makes, and for the same reason: the Core has no `!==` + // primitive (a source `!==` lowers to `not(core.eq(...))`), so without this the printer + // would otherwise hand `-D warnings` a `clippy::nonminimal_bool` finding on every one. + if (expr.op === "!" && expr.operand.kind === "method" && expr.operand.args.length === 0) { + if (expr.operand.name === "is_none") return `${print(expr.operand.target)}.is_some()`; + if (expr.operand.name === "is_some") return `${print(expr.operand.target)}.is_none()`; + } + if (expr.op === "!" && expr.operand.kind === "binary" && expr.operand.op === "==") { + return `(${print(expr.operand.left)} != ${print(expr.operand.right)})`; + } + const operand = print(expr.operand); + // `binary` already parenthesizes itself, so wrapping it again would double up. + const selfWrapped = + expr.operand.kind === "name" || + expr.operand.kind === "call" || + expr.operand.kind === "method" || + expr.operand.kind === "binary" || + expr.operand.kind === "index" || + expr.operand.kind === "member"; + return selfWrapped ? `${expr.op}${operand}` : `${expr.op}(${operand})`; + } + case "ternary": + // An `if`/`else` expression needs the same type on both arms, which a plain `print` + // cannot promise (a literal branch prints as `&str`, a computed one as `String`) without + // knowing the ternary's own type, which this function does not carry. `toOwned` with no + // expected type still converts a bare name or literal, and leaves an already-owned + // branch as it was, which is what reconciles the two sides for every case this backend + // actually emits (a `Copy` branch is untouched either way — see `toOwned`'s own comment). + return `(if ${print(expr.test)} { ${toOwned(expr.then, undefined)} } else { ${toOwned(expr.otherwise, undefined)} })`; + case "list": { + const elem = expr.type.kind === "List" ? expr.type.elem : undefined; + return `vec![${expr.items.map((item) => toOwned(item, elem)).join(", ")}]`; + } + case "record": { + const fields = recordFieldTypes.get(expr.typeName); + const rendered = expr.fields + .map((field) => `${field.name}: ${toOwned(field.value, fields?.get(fieldSourceName(field.name)))}`) + .join(", "); + const qualified = recordFieldTypes.has(expr.typeName) && isBuiltinRecord(expr.typeName); + return `${qualified ? `crate::support::${expr.typeName}` : expr.typeName} { ${rendered} }`; + } + case "lambda": { + const params = expr.params + .map((param) => `${isCopyType(param.type) ? `&${param.name}` : param.name}: &${rustType(param.type)}`) + .join(", "); + const lambdaBody = withFnCtx({ ret: expr.ret, fails: [] }, () => printBody(expr.body, paramScope(expr.params))); + return `|${params}| {\n${lambdaBody}\n}`; + } + case "none": + return "None"; + case "zero": + return "Default::default()"; + case "some": + // `toOwned` with no expected type still converts a bare name or literal correctly (see + // its own comment); `Some(...)` always needs an owned value, so this never wants a plain + // `print` the way most of this function's other cases do. + return `Some(${toOwned(expr.inner, undefined)})`; + default: { + const exhaustive: never = expr; + return exhaustive; + } + } +} + +const EMPTY_SCOPE: ReadonlyMap = new Map(); + +/** + * The scope `print` sees when it reaches a nested "call" node's arguments. Kept as module state, + * not a `print` parameter, so `print` itself keeps the single-argument shape a candidate's `emit` + * calls at lowering time (before any function's scope exists — `print` never needs one, since a + * candidate's own fragments never construct a "call" node; see the module comment). `printBody` + * sets this to the real scope for every statement it prints and restores it on the way out, so a + * "call" reached while actually printing a function body resolves its arguments' types correctly. + */ +let currentScope: ReadonlyMap = EMPTY_SCOPE; + +function paramScope(params: readonly { readonly name: string; readonly type: SemType }[]): Map { + return new Map(params.map((param) => [param.name, param.type])); +} + +/** A `TRecord` field name is already snake_cased; a builtin record's own field name is not. */ +function fieldSourceName(name: string): string { + return name; +} + +function isBuiltinRecord(name: string): boolean { + return BUILTIN_RECORDS.some((record) => record.name === name); +} + +/** `x == ""` / `x != ""`, either operand order, as `x.is_empty()` / `!x.is_empty()`. */ +function emptyStringCompare(expr: Extract): string | undefined { + const isEmpty = (e: TExpr): boolean => e.kind === "lit" && typeof e.value === "string" && e.value === ""; + const other = isEmpty(expr.left) ? expr.right : isEmpty(expr.right) ? expr.left : undefined; + if (other === undefined) return undefined; + const negated = expr.op === "!=" || expr.op === "!=="; + return `${negated ? "!" : ""}${print(other)}.is_empty()`; +} + +/** `x >= lo && x <= hi` (and the `||`-negated shape) as a `RangeInclusive`/`Range::contains`. */ +function rangeContains(expr: Extract): string | undefined { + if (expr.left.kind !== "binary" || expr.right.kind !== "binary") return undefined; + const left = expr.left; + const right = expr.right; + if (left.left.kind !== "name" || right.left.kind !== "name" || left.left.name !== right.left.name) { + return undefined; + } + const name = left.left.name; + if (expr.op === "&&" && left.op === ">=" && right.op === "<=") { + return `(${print(left.right)}..=${print(right.right)}).contains(&${name})`; + } + if (expr.op === "&&" && left.op === ">=" && right.op === "<") { + return `(${print(left.right)}..${print(right.right)}).contains(&${name})`; + } + if (expr.op === "||" && left.op === "<" && right.op === ">") { + return `!(${print(left.right)}..=${print(right.right)}).contains(&${name})`; + } + return undefined; +} + +function rustOperator(op: string): string { + switch (op) { + case "===": + return "=="; + case "!==": + return "!="; + default: + return op; + } +} + +function literal(value: Value, type?: SemType): string { + if (typeof value === "bigint") return value.toString(); + // Bare, not `.to_string()`: a string literal is already `&'static str`-compatible, which is + // what every borrow context here wants (and a `match` pattern requires — `"a".to_string()` is + // not a constant pattern at all); `toOwned`'s own "lit" case adds `.to_string()` wherever an + // owned value is actually needed, before ever reaching this function. + if (typeof value === "string") return rustString(value); + if (typeof value === "boolean") return String(value); + if (typeof value === "number") return String(value); + if (Array.isArray(value)) { + const elem = type !== undefined && type.kind === "List" ? type.elem : undefined; + return `vec![${value.map((item) => literal(item, elem)).join(", ")}]`; + } + return "None"; +} + +/* ------------------------------------------------------------------ * + * Statements + * ------------------------------------------------------------------ */ + +type Scope = Map; + +function printBody(body: readonly TStmt[], scope: Scope): string { + const previousScope = currentScope; + currentScope = scope; + try { + return printBodyLines(body, scope); + } finally { + currentScope = previousScope; + } +} + +function printBodyLines(body: readonly TStmt[], scope: Scope): string { + const lines: string[] = []; + for (let i = 0; i < body.length; i++) { + const statement = body[i]!; + // The shared lowerer hoists a fallible call into `let (v, err) = call; if err != nil { ... }` + // for a Go-shaped `errorsAsValues` target; here it collapses back into `let v = call()?;`, + // which is what makes `?` — the idiomatic form — come out of the same shared nanopass. + if (statement.kind === "multiLet" && statement.names.length === 2 && body[i + 1]?.kind === "if") { + const [value] = statement.names; + scope.set(value!, statement.types[0]!); + lines.push(`let ${value} = ${print(statement.init)}?;`); + i += 1; // the paired `if err != nil { return zero, err }` is now the `?` itself + continue; + } + lines.push(printStmt(statement, scope)); + } + return lines.join("\n"); +} + +function printStmt(statement: TStmt, scope: Scope): string { + switch (statement.kind) { + case "let": + scope.set(statement.name, statement.type); + return `let ${statement.mutable ? "mut " : ""}${statement.name} = ${toOwned(statement.init, statement.type)};`; + case "multiLet": + // Only reached with a name count other than 2, which the shared lowerer never produces. + return `let (${statement.names.join(", ")}) = ${print(statement.init)};`; + case "assign": { + // `x = x + step` is how the Core always expresses an accumulator update (there is no + // `+=` in the source language either); printing it as Rust's own compound assignment is + // what `clippy::assign_op_pattern` (default warn) otherwise asks for on every one. + const compound = + statement.target.kind === "name" && + statement.value.kind === "binary" && + "+-*/%".includes(statement.value.op) && + statement.value.left.kind === "name" && + statement.value.left.name === statement.target.name; + if (compound && statement.value.kind === "binary") { + return `${statement.target.name} ${statement.value.op}= ${print(statement.value.right)};`; + } + return `${print(statement.target)} = ${toOwned(statement.value, resolveType(statement.target, scope))};`; + } + case "if": { + const then = printBody(statement.then, new Map(scope)); + const otherwise = statement.otherwise.length === 0 ? "" : ` else {\n${printBody(statement.otherwise, new Map(scope))}\n}`; + return `if ${print(statement.test)} {\n${then}\n}${otherwise}`; + } + case "switch": { + const cases = statement.cases + .map( + (entry) => + `${entry.values.map((value) => literal(value)).join(" | ")} => {\n${printBody(entry.body, new Map(scope))}\n}`, + ) + .join("\n"); + const fallback = + statement.otherwise === undefined + ? `_ => unreachable!("the checker proves this switch exhaustive")` + : `_ => {\n${printBody(statement.otherwise, new Map(scope))}\n}`; + return `match ${print(statement.subject)} {\n${cases}\n${fallback}\n}`; + } + case "for": { + const inner = new Map(scope); + inner.set(statement.name, statement.type); + const body = printBody(statement.body, inner); + // The overwhelmingly common shape — ascending, step 1 — is an idiomatic Rust range; any + // other step or direction (never produced by this project today) falls back to a plain + // `while`, since Rust has no C-style `for` to carry an arbitrary step natively. + if (statement.step === 1n && !statement.inclusive) { + // A counted loop kept only for its trip count (`for (let attempt = 0; …)`, the + // redraw-bounding pattern `docs/semantics.md` names) never reads its own counter, + // which `unused_variables` (default warn) catches; `_name` is the same loop with the + // warning it would otherwise need suppressing. + const binding = new RegExp(`\\b${statement.name}\\b`).test(body) ? statement.name : `_${statement.name}`; + return `for ${binding} in ${print(statement.from)}..${print(statement.to)} {\n${body}\n}`; + } + const forward = statement.step > 0n; + const cmp = forward ? (statement.inclusive ? "<=" : "<") : statement.inclusive ? ">=" : ">"; + const update = statement.step === 1n ? "+= 1" : statement.step === -1n ? "-= 1" : `+= ${statement.step}`; + return [ + "{", + `\tlet mut ${statement.name}: i64 = ${print(statement.from)};`, + `\twhile ${statement.name} ${cmp} ${print(statement.to)} {`, + indent(body, 2), + `\t\t${statement.name} ${update};`, + "\t}", + "}", + ].join("\n"); + } + case "forEach": { + const inner = new Map(scope); + inner.set(statement.name, statement.type); + const body = printBody(statement.body, inner); + const binding = isCopyType(statement.type) ? `&${statement.name}` : statement.name; + return `for ${binding} in ${print(statement.iterable)}.iter() {\n${body}\n}`; + } + case "return": { + if (statement.value === undefined) { + return currentFnCtx.fails.length === 0 ? "return;" : "return Ok(());"; + } + const rendered = toOwned(statement.value, currentFnCtx.ret); + return `return ${currentFnCtx.fails.length > 0 ? `Ok(${rendered})` : rendered};`; + } + case "throw": { + const message = statement.args[0] === undefined ? `"".to_string()` : toOwned(statement.args[0], undefined); + return `return Err(CoreError::${statement.errorClass} { message: ${message} });`; + } + case "break": + return "break;"; + case "continue": + return "continue;"; + case "expr": + return `${print(statement.expr)};`; + case "raw": + return statement.text; + default: { + const exhaustive: never = statement; + return exhaustive; + } + } +} + +/** + * The enclosing function's (or lambda's) return type and fallibility, for `return` alone — the + * one statement whose rendering depends on something outside the statement tree itself. Module + * state rather than a threaded parameter, for the same reason `currentScope` is: it lets `print` + * stay a single-argument function. `printFunction` and the `lambda` case of `print` are the only + * two places that set it (a lambda is never itself fallible, so it always installs `fails: []`). + */ +let currentFnCtx: { readonly ret: SemType; readonly fails: readonly string[] } = { ret: { kind: "Void" }, fails: [] }; + +function withFnCtx(ctx: { readonly ret: SemType; readonly fails: readonly string[] }, run: () => T): T { + const previous = currentFnCtx; + currentFnCtx = ctx; + try { + return run(); + } finally { + currentFnCtx = previous; + } +} + +function indent(text: string, levels: number): string { + const pad = "\t".repeat(levels); + return text + .split("\n") + .map((line) => (line === "" ? line : `${pad}${line}`)) + .join("\n"); +} + +/** + * A parameter's declared type: `rustType` for every parameter but the ones the borrow pre-pass + * (`analysis/borrows.ts`) found read-only, which print as a reference instead of an owned value — + * `TParam.borrowed` is this function's only input, consulted rather than decided here, exactly the + * way `docs/decisions/0010-*.md` describes. Only `String`/`List` have that distinction to make; a + * `Capabilities` environment is handled separately, by its own always-borrowed case below. + */ +function rustParamType(type: SemType, borrowed: boolean): string { + if (!borrowed) return rustType(type); + if (type.kind === "String" || type.kind === "Enum") return "&str"; + if (type.kind === "List") return `&[${rustType(type.elem)}]`; + return rustType(type); +} + +export function printFunction(fn: TFunc): string { + const scope = paramScope(fn.params.map((param) => ({ name: param.name, type: param.type }))); + const params = fn.params + .map((param) => { + if (param.type.kind === "Record" && param.type.name === "Capabilities") return `${param.name}: &dyn Capabilities`; + return `${param.name}: ${rustParamType(param.type, param.borrowed === true)}`; + }) + .join(", "); + const returnType = fn.fails.length > 0 ? `Result<${rustType(fn.ret)}, CoreError>` : rustType(fn.ret); + const doc = fn.doc === undefined ? "" : `${fn.doc.split("\n").map((line) => `/// ${line}`.trimEnd()).join("\n")}\n`; + const body = withFnCtx({ ret: fn.ret, fails: fn.fails }, () => printBody(fn.body, scope)); + // `exported` (a utility) is `pub`, reachable from outside the crate through `lib.rs`'s flat + // `pub use module::*;` re-export. `moduleExported` alone — the source module's own `export`, + // which a library helper carries so other generated modules can call it — is `pub(crate)`: a + // glob re-export silently drops an item that isn't at least as visible as the `pub use` itself + // (verified against rustc directly), so this never leaks a library helper past the crate the + // way a plain `pub` would. Neither flag set means the source never exported it at all, and + // every caller found in `docs/semantics.md`'s survey lives in the same module, so a bare `fn` + // — visible only here — is enough. + const visibility = fn.exported ? "pub " : fn.moduleExported ? "pub(crate) " : ""; + return `${doc}${visibility}fn ${fn.name}(${params}) -> ${returnType} {\n${indent(body, 1)}\n}`; +} + +export function printRecord(record: TRecord): string { + const doc = record.doc === undefined ? "" : `/// ${record.doc.split("\n")[0]}\n`; + const fields = record.fields.map((field) => `\tpub ${field.name}: ${rustType(field.type)},`).join("\n"); + return `${doc}#[derive(Clone, Debug, PartialEq)]\npub struct ${record.name} {\n${fields}\n}`; +} + +/** A hoisted constant table: always a list literal of a Copy element type (see `hoistConstantTables`). */ +function printConstant(name: string, value: TExpr): string { + const elem = value.kind === "lit" && Array.isArray(value.value) && value.type.kind === "List" ? value.type.elem : undefined; + const items = value.kind === "lit" && Array.isArray(value.value) ? value.value : []; + const rendered = items.map((item) => literal(item, elem)).join(", "); + return `pub const ${name.toUpperCase()}: &[${elem === undefined ? "i64" : rustType(elem)}] = &[${rendered}];`; +} + +export function printModule(module: TModule): string { + rebuildRecordFieldTypes(module.records); + hoistedConstantNames.clear(); + hoistedConstantTypes.clear(); + for (const constant of module.constants) { + hoistedConstantNames.add(constant.name); + hoistedConstantTypes.set(constant.name, constant.type); + } + const rendered = [ + ...module.records.map(printRecord), + ...module.constants.map((constant) => printConstant(constant.name, constant.value)), + ...module.functions.map(printFunction), + ].join("\n\n"); + // A plain function call is never qualified at its call site (see the "call" case of `print`), + // so every module needs every other module's public items in scope; `lib.rs` re-exports each + // module flatly (`pub use lib_digits::*;` and so on) for this glob to resolve against. Every + // module gets it unconditionally rather than only the ones `computeImports` says reach outside + // themselves, because that signal also fires for a portable or capability-only dependency this + // file already qualifies explicitly (`crate::std_date::...`, `&dyn Capabilities`) — a real but + // imprecise "might not end up using the glob" case that `unused_imports` would otherwise catch; + // `lib.rs`'s crate-level `#![allow(unused_imports)]` is what that imprecision costs. + const glob = "use crate::*;\n\n"; + return `${module.header}\n\n${glob}${rendered}\n`; +} + +function importPath(_from: string, to: string): string { + return rustModuleName(to); +} + +/* ------------------------------------------------------------------ * + * Support file: the capability trait, the JSON-free helpers the capability table names, and the + * hand-written regex matcher. + * ------------------------------------------------------------------ */ + +/** + * The fallback matcher, for a pattern `chainElementsOf` (in the "Regex" section above) refused. It + * is a direct port of \`regex.ts\`'s own reference matcher (\`matchNode\`): continuation-passing + * backtracking over \`&str\` byte slices, greedy repeats trying one more repetition before giving up + * and continuing. Porting that exact algorithm, rather than a from-scratch one, is what makes this + * side sound without a separate proof: it is already what the engine's own tests check every + * accepted pattern against. + * + * There is no allocation anywhere in it. \`Class\` borrows a suffix of its input; \`Seq\` and + * \`Repeat\` build their continuations as stack-local closures passed by reference (\`&dyn Fn\`, + * never \`Box\`), so backtracking costs stack frames, not heap traffic — the "Vec and a + * per-node Vec on every call" problem this whole fix exists to remove never had to be + * replaced with a differently-shaped allocation, because CPS over slices does not need one. + */ +const RE_MATCHER_SUPPORT = ` +/// One node of a normalized regex, compiled at engine build time into a \`static\` value: every +/// child is a \`&'static\` reference to a const expression, so rustc places the whole tree in the +/// binary's read-only data once. There is no run-time compilation step to avoid. +pub enum ReNode { + Class(&'static [(u32, u32)], bool), + Seq(&'static [ReNode]), + Alt(&'static [ReNode]), + Repeat(&'static ReNode, usize, Option), +} + +fn re_class_matches(ranges: &[(u32, u32)], negated: bool, scalar: u32) -> bool { + let hit = ranges.iter().any(|&(lo, hi)| scalar >= lo && scalar <= hi); + hit != negated +} + +/// Matches \`node\` at the front of \`rest\`, then hands whatever remains to \`cont\`; answers true for +/// the first way through \`node\` (greedy branch first) whose continuation also accepts. \`rest\` is +/// always a UTF-8 boundary slice of the original input, so every step is a borrow, never a copy. +fn re_match<'a>(node: &'static ReNode, rest: &'a str, cont: &dyn Fn(&'a str) -> bool) -> bool { + match node { + ReNode::Class(ranges, negated) => match rest.chars().next() { + Some(c) if re_class_matches(ranges, *negated, c as u32) => cont(&rest[c.len_utf8()..]), + _ => false, + }, + ReNode::Seq(items) => re_match_seq(items, rest, cont), + ReNode::Alt(options) => options.iter().any(|option| re_match(option, rest, cont)), + ReNode::Repeat(item, min, max) => re_match_repeat(item, *min, max.unwrap_or(usize::MAX), 0, rest, cont), + } +} + +fn re_match_seq<'a>(items: &'static [ReNode], rest: &'a str, cont: &dyn Fn(&'a str) -> bool) -> bool { + match items.split_first() { + None => cont(rest), + Some((first, remaining)) => { + let next_cont = move |next: &'a str| re_match_seq(remaining, next, cont); + re_match(first, rest, &next_cont) + } + } +} + +fn re_match_repeat<'a>( + item: &'static ReNode, + min: usize, + limit: usize, + count: usize, + rest: &'a str, + cont: &dyn Fn(&'a str) -> bool, +) -> bool { + if count < limit { + let rest_len = rest.len(); + // A zero-width repetition would loop forever; the accepted subset never needs one (an + // empty repeated item is rejected up front), so the length check is just that guard. + let advance_cont = + move |next: &'a str| next.len() != rest_len && re_match_repeat(item, min, limit, count + 1, next, cont); + if re_match(item, rest, &advance_cont) { + return true; + } + } + count >= min && cont(rest) +} + +/// Whether \`value\` fully matches \`pattern\`, anchored at both ends (the only mode the Core admits). +pub fn re_test(pattern: &'static ReNode, value: &str) -> bool { + re_match(pattern, value, &|rest| rest.is_empty()) +} +`; + +// Every helper below takes its arguments borrowed, not owned — unlike a generated project +// function (see the module comment on ownership), which is why every call site in the capability +// table borrows explicitly (`&value`, never bare `value`): a reference is `Copy`, so the same +// local can be handed to as many of these calls as an operation needs, in a loop or anywhere +// else, without the caller ever losing it. Project functions cannot do the same for one another in +// general (an owned parameter may need to be stored, not just read), but every one of *these* +// helpers only ever reads. +const GENERIC_SUPPORT = ` +pub fn concat2(a: &str, b: &str) -> String { + let mut out = String::with_capacity(a.len() + b.len()); + out.push_str(a); + out.push_str(b); + out +} + +pub fn code_points(value: &str) -> Vec { + value.chars().map(|c| c as i64).collect() +} + +pub fn from_code_points(points: &[i64]) -> String { + points + .iter() + .map(|&p| char::from_u32(p as u32).unwrap_or('\\u{fffd}')) + .collect() +} + +pub fn as_ascii(value: &str) -> Option { + if value.chars().all(|c| (c as u32) < 0x80) { + Some(value.to_string()) + } else { + None + } +} + +pub fn as_digits(value: &str) -> Option { + if !value.is_empty() && value.chars().all(|c| c.is_ascii_digit()) { + Some(value.to_string()) + } else { + None + } +} + +pub fn parse_digits(value: &str) -> Option { + if value.is_empty() || value.len() > 18 || !value.chars().all(|c| c.is_ascii_digit()) { + return None; + } + value.parse::().ok() +} + +/// Consumes exactly \`count\` chars matching \`in_class\` off the front of \`rest\`, or answers \`None\` +/// without consuming anything. One forward pass, no allocation: this and \`re_take_class\` below are +/// the whole of a generated chain-pattern scanner (\`re_match_N\`, in the "Regex" section of +/// engine/src/targets/rust/index.ts) — a fixed-count class run in the pattern becomes one call here. +/// Every class this project matches against is ASCII except the mask-separator whitespace class, +/// and even that one is ASCII on almost every byte a real caller passes (plain digits, or digits +/// plus '.', '-', '/' and ' ' -- see \`core/source\`'s own patterns) -- so the leading byte is tested +/// directly first; only a byte that starts a multi-byte sequence pays for decoding a full \`char\`. +#[inline] +fn re_take_fixed(rest: &str, count: usize, in_class: impl Fn(u32) -> bool) -> Option<&str> { + let bytes = rest.as_bytes(); + let mut pos = 0usize; + let mut taken = 0usize; + while taken < count { + let &b = bytes.get(pos)?; + if b < 0x80 { + if !in_class(b as u32) { + return None; + } + pos += 1; + } else { + let ch = rest[pos..].chars().next().unwrap(); + if !in_class(ch as u32) { + return None; + } + pos += ch.len_utf8(); + } + taken += 1; + } + Some(&rest[pos..]) +} + +/// Consumes as many chars matching \`in_class\` as \`rest\` offers, up to \`max\` (\`usize::MAX\` for +/// unbounded), then answers \`None\` unless at least \`min\` were taken. The maximal-munch property +/// \`chainElementsOf\` checks at generation time (see the "Regex" section) is what makes always +/// taking the longest available run — never backing off to try a shorter one — correct here. Same +/// ASCII-first byte test as \`re_take_fixed\` above, for the same reason. +#[inline] +fn re_take_class(rest: &str, min: usize, max: usize, in_class: impl Fn(u32) -> bool) -> Option<&str> { + let bytes = rest.as_bytes(); + let mut pos = 0usize; + let mut taken = 0usize; + while taken < max { + let Some(&b) = bytes.get(pos) else { + break; + }; + if b < 0x80 { + if !in_class(b as u32) { + break; + } + pos += 1; + } else { + let ch = rest[pos..].chars().next().unwrap(); + if !in_class(ch as u32) { + break; + } + pos += ch.len_utf8(); + } + taken += 1; + } + if taken < min { + return None; + } + Some(&rest[pos..]) +} + +/// The single-digit fast path \`str.fromInt\` prefers when the value is proven to be one decimal +/// digit (see the candidate's own comment in engine/src/targets/rust/index.ts): one ASCII byte +/// pushed into a one-byte-capacity \`String\` is the whole job, no general integer formatter needed. +pub fn digit_char(n: i64) -> String { + let mut out = String::with_capacity(1); + out.push((n as u8 + b'0') as char); + out +} + +/// The ASCII-only fast path \`str.padStart\` prefers when both \`value\` and \`pad\` are proven ASCII +/// (see the candidate's own comment in engine/src/targets/rust/index.ts): a scalar is a byte, so +/// the length check is \`value.len()\` and each missing slot is \`pad\` pushed wholesale, with no +/// \`Vec\` built anywhere -- \`pad_start\` below builds two just to learn what this already +/// knows. +pub fn pad_start_ascii(value: &str, length: i64, pad: &str) -> String { + let length = length as usize; + if value.len() >= length { + return value.to_string(); + } + let missing = length - value.len(); + let mut out = String::with_capacity(pad.len() * missing + value.len()); + for _ in 0..missing { + out.push_str(pad); + } + out.push_str(value); + out +} + +pub fn pad_start(value: &str, length: i64, pad: &str) -> String { + let scalars: Vec = value.chars().collect(); + let length = length as usize; + if scalars.len() >= length { + return value.to_string(); + } + let pad_scalars: Vec = pad.chars().collect(); + let mut prefix = String::new(); + for _ in 0..(length - scalars.len()) { + prefix.extend(pad_scalars.iter()); + } + prefix + value +} + +pub fn code_at(value: &str, index: i64) -> Option { + let bytes = value.as_bytes(); + if index < 0 || index as usize >= bytes.len() { + return None; + } + Some(bytes[index as usize] as i64) +} + +pub fn char_at(value: &str, index: i64) -> Option { + let bytes = value.as_bytes(); + if index < 0 || index as usize >= bytes.len() { + return None; + } + Some((bytes[index as usize] as char).to_string()) +} + +pub fn at(values: &[T], index: i64) -> Option { + if index < 0 { + return None; + } + values.get(index as usize).cloned() +} + +pub fn get_at(values: &[T], index: i64) -> T { + values[index as usize].clone() +} + +pub fn mapped(values: &[T], f: impl Fn(&T) -> R) -> Vec { + values.iter().map(f).collect() +} + +pub fn filtered(values: &[T], keep: impl Fn(&T) -> bool) -> Vec { + values.iter().filter(|v| keep(v)).cloned().collect() +} + +pub fn sorted_stable(values: &[T]) -> Vec { + let mut out = values.to_vec(); + out.sort(); + out +} + +pub fn sorted_stable_by(values: &[T], key: impl Fn(&T) -> K) -> Vec { + let mut out = values.to_vec(); + out.sort_by_key(key); + out +} + +pub fn date_from_epoch_days(days: i64) -> Option { + if !(-719162..=2932896).contains(&days) { + return None; + } + Some(days) +} +`; + +const RACE_SUPPORT = ` +/// Runs each task on its own thread and answers the first one that lands \`Some\`. Cancellation is +/// best effort and semantically unobservable, exactly as \`docs/semantics.md\` describes: a losing +/// task may run to completion, and its answer is dropped on the floor. Tasks are matched to +/// completion order through a channel, not through joining threads in task order, which is what +/// makes this "the first task that answers" rather than "the first task in the list". +pub fn race_first_some<'scope, T: Send + 'scope>( + tasks: Vec Option + Send + 'scope>>, +) -> Option { + let count = tasks.len(); + let (sender, receiver) = std::sync::mpsc::channel::>(); + std::thread::scope(|scope| { + for task in tasks { + let sender = sender.clone(); + scope.spawn(move || { + let _ = sender.send(task()); + }); + } + drop(sender); + for _ in 0..count { + if let Ok(Some(value)) = receiver.recv() { + return Some(value); + } + } + None + }) +} +`; + +const CAPABILITIES_SUPPORT = ` +/// One request or response header. Headers are an ordered list, never a map, so every target +/// preserves order and duplicates. +#[derive(Clone, Debug, PartialEq)] +pub struct HttpHeader { + pub name: String, + pub value: String, +} + +/// A request handed to the Http capability. The host adds no retries and no hidden headers. +#[derive(Clone, Debug, PartialEq)] +pub struct HttpRequest { + pub method: String, + pub url: String, + pub headers: Vec, + pub body: String, + pub timeout_millis: i64, +} + +/// A response from the Http capability. A status of 400 or more is a value, not a failure. +#[derive(Clone, Debug, PartialEq)] +pub struct HttpResponse { + pub status: i64, + pub headers: Vec, + pub body: String, +} + +/// Everything the generated core needs from the outside world. \`std\` has neither an HTTP client +/// nor a source of randomness or wall-clock time in one place, so the core takes this trait and +/// leaves the default implementation to the host — the same split every other capability +/// intrinsic uses, and the first friction \`docs/targets/rust-sketch.md\` predicted. +/// +/// \`task.race\` spawns one thread per task (see \`race_first_some\` above), so an implementation has +/// to be safe to share across threads; \`Sync\` is what that costs a hand-written implementation +/// that \`std\` alone cannot supply. +pub trait Capabilities: Sync { + /// A transport error or a timeout answers \`None\`; a 4xx or 5xx status is a value. + fn request(&self, request: HttpRequest) -> Option; + fn now(&self) -> i64; + fn sleep(&self, millis: i64); + fn next_u32(&self) -> i64; +} +`; + +function supportModule(_program: CProgram, needs: SupportNeeds): { path: string; text: string } { + const parts = [ + "// Code generated by the logic engine. DO NOT EDIT.", + `// engine: ${ENGINE_VERSION}`, + "// source: support", + "#![allow(dead_code)]", + "", + GENERIC_SUPPORT.trim(), + ]; + // `RE_MATCHER_SUPPORT` backs only the fallback path (see the "Regex" section above); a project + // whose patterns all take the dedicated-scanner path, like this one, never needs it, and + // `#![allow(dead_code)]` above does not excuse shipping a matcher nothing calls. + if (reFallbackUsed) parts.push("", RE_MATCHER_SUPPORT.trim()); + parts.push("", ...[...rePatterns.values()].map((pattern) => pattern.rust)); + if (needs.race) parts.push("", RACE_SUPPORT.trim()); + if (needs.env) parts.push("", CAPABILITIES_SUPPORT.trim()); + return { path: "src/support.rs", text: `${parts.join("\n")}\n` }; +} + +function errorsModule(program: CProgram): { path: string; text: string } | undefined { + const declared = [...program.errors.values()]; + if (declared.length === 0) return undefined; + const lines = [ + "// Code generated by the logic engine. DO NOT EDIT.", + `// engine: ${ENGINE_VERSION}`, + "// source: errors", + "", + "/// Every domain error the project declares, as one flat enum: a nested fallible call is", + "/// `f(...)?` because every fallible function shares this one error type, with never a", + "/// mismatch for `?` to bridge. The sketch expected one type per utility; this is simpler,", + "/// and loses nothing `errors.Is`-shaped, since the variant itself is the family membership", + "/// test (`matches!(err, CoreError::SomeVariant { .. })`).", + "#[derive(Clone, Debug, PartialEq)]", + "pub enum CoreError {", + ...declared.map((error) => `\t${error.name} { message: String },`), + "}", + "", + "impl std::fmt::Display for CoreError {", + "\tfn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {", + "\t\tmatch self {", + ...declared.map((error) => `\t\t\tCoreError::${error.name} { message } => write!(f, "{}", message),`), + "\t\t}", + "\t}", + "}", + "", + "impl std::error::Error for CoreError {}", + "", + ]; + return { path: "src/errors.rs", text: lines.join("\n") }; +} + +/* ------------------------------------------------------------------ * + * Driver: `Cargo.toml`, `src/lib.rs` and the differential driver binary. + * + * The driver speaks the same `{"fn":...,"args":[...]}` / `{"ok":...}` line protocol as every + * other target. `std` has no JSON, so this file hand-writes a minimal reader and writer for it — + * driver code, not project code, in the same sense the regex matcher above is: nothing here is a + * dependency, it is the one piece of machinery the differential harness needs that `std` does not + * supply. + * ------------------------------------------------------------------ */ + +const emittedModules: { path: string }[] = []; + +const JSON_SUPPORT = ` +//! A minimal JSON reader and writer for the differential driver's own protocol. This is driver +//! code, not project code: the driver has to speak JSON lines to the conformance harness, and +//! \`std\` has none, so this is the one piece of machinery it needs that \`std\` does not supply — +//! not a general-purpose JSON library, just enough to round-trip the protocol's own shapes +//! (numbers, strings, bools, null, arrays and objects of those). + +#[derive(Clone, Debug, PartialEq)] +pub enum Json { + Null, + Bool(bool), + Number(f64), + String(String), + Array(Vec), + Object(Vec<(String, Json)>), +} + +impl Json { + pub fn as_str(&self) -> &str { + match self { + Json::String(s) => s, + _ => "", + } + } + + pub fn as_i64(&self) -> i64 { + match self { + Json::Number(n) => *n as i64, + _ => 0, + } + } + + pub fn as_f64(&self) -> f64 { + match self { + Json::Number(n) => *n, + _ => 0.0, + } + } + + pub fn as_bool(&self) -> bool { + matches!(self, Json::Bool(true)) + } + + pub fn as_array(&self) -> &[Json] { + match self { + Json::Array(items) => items, + _ => &[], + } + } + + pub fn get(&self, key: &str) -> Option<&Json> { + match self { + Json::Object(fields) => fields.iter().find(|(name, _)| name == key).map(|(_, value)| value), + _ => None, + } + } +} + +pub fn parse(text: &str) -> Json { + let chars: Vec = text.chars().collect(); + let mut pos = 0usize; + parse_value(&chars, &mut pos) +} + +fn skip_space(chars: &[char], pos: &mut usize) { + while *pos < chars.len() && chars[*pos].is_whitespace() { + *pos += 1; + } +} + +fn parse_value(chars: &[char], pos: &mut usize) -> Json { + skip_space(chars, pos); + match chars.get(*pos) { + Some('{') => parse_object(chars, pos), + Some('[') => parse_array(chars, pos), + Some('"') => Json::String(parse_string(chars, pos)), + Some('t') => { + *pos += 4; + Json::Bool(true) + } + Some('f') => { + *pos += 5; + Json::Bool(false) + } + Some('n') => { + *pos += 4; + Json::Null + } + _ => parse_number(chars, pos), + } +} + +fn parse_object(chars: &[char], pos: &mut usize) -> Json { + *pos += 1; + let mut fields = Vec::new(); + skip_space(chars, pos); + if chars.get(*pos) == Some(&'}') { + *pos += 1; + return Json::Object(fields); + } + loop { + skip_space(chars, pos); + let key = parse_string(chars, pos); + skip_space(chars, pos); + *pos += 1; // ':' + let value = parse_value(chars, pos); + fields.push((key, value)); + skip_space(chars, pos); + match chars.get(*pos) { + Some(',') => { + *pos += 1; + } + _ => { + *pos += 1; // '}' + break; + } + } + } + Json::Object(fields) +} + +fn parse_array(chars: &[char], pos: &mut usize) -> Json { + *pos += 1; + let mut items = Vec::new(); + skip_space(chars, pos); + if chars.get(*pos) == Some(&']') { + *pos += 1; + return Json::Array(items); + } + loop { + let value = parse_value(chars, pos); + items.push(value); + skip_space(chars, pos); + match chars.get(*pos) { + Some(',') => { + *pos += 1; + } + _ => { + *pos += 1; // ']' + break; + } + } + } + Json::Array(items) +} + +fn parse_string(chars: &[char], pos: &mut usize) -> String { + *pos += 1; // opening quote + let mut out = String::new(); + while let Some(&c) = chars.get(*pos) { + *pos += 1; + if c == '"' { + break; + } + if c == '\\\\' { + let escaped = chars.get(*pos).copied().unwrap_or('\\\\'); + *pos += 1; + match escaped { + 'n' => out.push('\\n'), + 'r' => out.push('\\r'), + 't' => out.push('\\t'), + 'u' => { + let hex: String = chars[*pos..*pos + 4].iter().collect(); + *pos += 4; + if let Ok(code) = u32::from_str_radix(&hex, 16) { + if let Some(scalar) = char::from_u32(code) { + out.push(scalar); + } + } + } + other => out.push(other), + } + } else { + out.push(c); + } + } + out +} + +fn parse_number(chars: &[char], pos: &mut usize) -> Json { + let start = *pos; + while chars + .get(*pos) + .is_some_and(|c| c.is_ascii_digit() || *c == '-' || *c == '+' || *c == '.' || *c == 'e' || *c == 'E') + { + *pos += 1; + } + let text: String = chars[start..*pos].iter().collect(); + Json::Number(text.parse::().unwrap_or(0.0)) +} + +pub fn write(value: &Json) -> String { + match value { + Json::Null => "null".to_string(), + Json::Bool(b) => b.to_string(), + Json::Number(n) => { + if n.fract() == 0.0 && n.abs() < 1e15 { + format!("{}", *n as i64) + } else { + format!("{}", n) + } + } + Json::String(s) => write_string(s), + Json::Array(items) => format!("[{}]", items.iter().map(write).collect::>().join(",")), + Json::Object(fields) => format!( + "{{{}}}", + fields + .iter() + .map(|(key, value)| format!("{}:{}", write_string(key), write(value))) + .collect::>() + .join(",") + ), + } +} + +fn write_string(value: &str) -> String { + let mut out = String::from("\\""); + for c in value.chars() { + match c { + '"' => out.push_str("\\\\\\""), + '\\\\' => out.push_str("\\\\\\\\"), + '\\n' => out.push_str("\\\\n"), + '\\r' => out.push_str("\\\\r"), + '\\t' => out.push_str("\\\\t"), + c if (c as u32) < 0x20 => out.push_str(&format!("\\\\u{:04x}", c as u32)), + c => out.push(c), + } + } + out.push('"'); + out +} +`; + +function decodeArg(type: SemType, index: number): string { + switch (type.kind) { + case "Bool": + return `args[${index}].as_bool()`; + case "Float": + return `args[${index}].as_f64()`; + case "Int": + case "Decimal": + case "CivilDate": + case "Instant": + case "Duration": + return `args[${index}].as_i64()`; + case "String": + case "Enum": + return `args[${index}].as_str().to_string()`; + case "List": { + const item = decodeJsonValue(type.elem, "item"); + return `args[${index}].as_array().iter().map(|item| ${item}).collect::>()`; + } + case "Record": { + const fields = decodeRecordFields(type.name, `args[${index}]`); + return `${pascal(type.name)} { ${fields} }`; + } + default: + return `args[${index}].as_str().to_string()`; + } +} + +function decodeJsonValue(type: SemType, varName: string): string { + switch (type.kind) { + case "Bool": + return `${varName}.as_bool()`; + case "Float": + return `${varName}.as_f64()`; + case "Int": + case "Decimal": + case "CivilDate": + case "Instant": + case "Duration": + return `${varName}.as_i64()`; + case "Record": { + const fields = decodeRecordFields(type.name, varName); + return `${pascal(type.name)} { ${fields} }`; + } + default: + return `${varName}.as_str().to_string()`; + } +} + +/** The wire name a keyword-escaped Rust field (`r#type`) was named before escaping (`type`). */ +function wireName(name: string): string { + return name.startsWith("r#") ? name.slice(2) : name; +} + +function decodeRecordFields(recordName: string, jsonVar: string): string { + const fields = recordFieldTypes.get(recordName); + if (fields === undefined) return ""; + return [...fields.entries()] + .map(([name, type]) => `${name}: ${decodeJsonValue(type, `${jsonVar}.get("${wireName(name)}").unwrap()`)}`) + .join(", "); +} + +function encodeValue(expr: string, type: SemType): string { + switch (type.kind) { + case "Bool": + return `coreout::json::Json::Bool(${expr})`; + case "Float": + return `coreout::json::Json::Number(${expr})`; + case "Int": + case "Decimal": + case "CivilDate": + case "Instant": + case "Duration": + return `coreout::json::Json::Number(${expr} as f64)`; + case "String": + case "Enum": + return `coreout::json::Json::String(${expr})`; + case "Option": { + const inner = encodeValue("inner", type.inner); + return `match ${expr} { Some(inner) => ${inner}, None => coreout::json::Json::Null }`; + } + case "List": { + const inner = encodeValue("item", type.elem); + return `coreout::json::Json::Array(${expr}.into_iter().map(|item| ${inner}).collect())`; + } + case "Record": { + const fields = recordFieldTypes.get(type.name); + const rendered = + fields === undefined + ? "" + : [...fields.entries()] + .map(([name, fieldType]) => `("${wireName(name)}".to_string(), ${encodeValue(`rec.${name}`, fieldType)})`) + .join(", "); + // A dedicated binding name (`rec`, not `value`): `expr` is sometimes literally `"value"` + // already (the outermost call in `driverFiles`), and rebinding it to itself is exactly + // what `clippy::redundant_locals` (default warn) exists to catch. + return `{ let rec = ${expr}; coreout::json::Json::Object(vec![${rendered}]) }`; + } + default: + return `coreout::json::Json::Null`; + } +} + +function driverFiles(program: CProgram, entries: readonly DriverEntry[]): { path: string; text: string }[] { + const modules = [...emittedModules]; + emittedModules.length = 0; + // The driver is hand-written-shaped generated code, outside the Target AST this file's own + // `print`/`callArg` consult for every other call site, so it borrows its own entry-point calls + // straight from the borrow map rather than through `borrowedArgs` (see `docs/decisions/0010-*.md`). + const borrows = computeBorrowableParams(program); + + const cases = entries.map((entry) => { + const callee = program.functions.get(entry.coreName); + const borrowedHere = borrows.get(entry.coreName); + const args = entry.params.map((type, index) => { + const paramName = callee?.params[index]?.name; + const wantsBorrow = paramName !== undefined && borrowedHere?.has(paramName) === true; + // A borrowed `String`/`Enum` parameter reads `args[i].as_str()` directly — already a + // `&str` into the decoded JSON value, so there is no owned `String` to borrow a reference + // to in the first place (`clippy::unnecessary_to_owned`, default warn, is what building + // one only to immediately `&`-reference it would trip). A borrowed `List` has no such + // borrowed JSON accessor to read instead, so it still decodes into an owned `Vec` and + // borrows that: `&decoded` on a freshly built temporary is ordinary Rust temporary + // lifetime extension, not a dangling reference. + if (wantsBorrow && (type.kind === "String" || type.kind === "Enum")) return `args[${index}].as_str()`; + const decoded = decodeArg(type, index); + return wantsBorrow ? `&${decoded}` : decoded; + }); + // `coreout`'s `lib.rs` re-exports every module flatly (see `driverFiles` below), and the + // driver brings that flat namespace in with `use coreout::*;`, so a bare name resolves. + // `dispatch`'s own `environment` parameter is already `&dyn Capabilities` (unlike + // `new_environment()`'s `Box`, which is where `.as_ref()` belongs). + const call = `${entry.targetName}(${[...args, ...(entry.usesEnv ? ["environment"] : [])].join(", ")})`; + const encode = encodeValue("value", entry.ret); + if (entry.fails.length > 0) { + return [ + `\t\t${JSON.stringify(entry.coreName)} => match ${call} {`, + `\t\t\tOk(value) => coreout::json::Json::Object(vec![("ok".to_string(), coreout::json::Json::Bool(true)), ("value".to_string(), ${encode})]),`, + `\t\t\tErr(err) => coreout::json::Json::Object(vec![("ok".to_string(), coreout::json::Json::Bool(false)), ("error".to_string(), coreout::json::Json::String(error_name(&err)))]),`, + "\t\t},", + ].join("\n"); + } + return [ + `\t\t${JSON.stringify(entry.coreName)} => {`, + `\t\t\tlet value = ${call};`, + `\t\t\tcoreout::json::Json::Object(vec![("ok".to_string(), coreout::json::Json::Bool(true)), ("value".to_string(), ${encode})])`, + "\t\t}", + ].join("\n"); + }); + + const usesEnv = entries.some((entry) => entry.usesEnv); + + const moduleNames = modules.map((module) => module.path.replace(/^src\//, "").replace(/\.rs$/, "")); + const libLines = [ + "// Code generated by the logic engine. DO NOT EDIT.", + `// engine: ${ENGINE_VERSION}`, + "// source: lib", + "//", + "// `needless_return`: every function body ends with an explicit `return`, mirroring the", + "// Core's own structure (see the module comment on ownership); a tail-position rewrite would", + "// have to reconstruct reachability through `if`/`match`/loops that the Core already settled.", + "// `unused_parens`: the printer parenthesizes every binary and ternary uniformly rather than", + "// tracking each operator's precedence, which is what makes the printer itself precedence-free.", + "// `vec_init_then_push`: `push` on a local list, one call per element, is the Core's own", + "// accumulation idiom (`docs/semantics.md` names it explicitly); collapsing a run of them into", + "// one `vec![...]` literal would need the printer to prove nothing between them can fail or", + "// branch, which a source author's control flow can defeat in ways this backend does not chase.", + "#![allow(clippy::needless_return, clippy::vec_init_then_push, unused_parens, unused_imports)]", + "", + "pub mod support;", + // A minimal JSON reader/writer for the driver's own protocol (see `JSON_SUPPORT`); it lives + // here, as a module of the library crate, only because Cargo would otherwise mistake a file + // under `src/bin/` for a binary target of its own. `driver.rs` is the only importer. + "pub mod json;", + ...(program.errors.size > 0 ? ["pub mod errors;"] : []), + ...moduleNames.map((name) => `pub mod ${name};`), + "", + // Flat re-exports: a module's own `use crate::*;` (see `printModule`) is how a cross-module + // call resolves without being qualified at its call site, and this is the flattening that + // makes it resolve. Every generated function's name is unique program-wide (the shared + // name assignment dedupes across the whole program, not per module), so this never collides. + "pub use support::*;", + ...(program.errors.size > 0 ? ["pub use errors::*;"] : []), + ...moduleNames.map((name) => `pub use ${name}::*;`), + "", + ]; + + // Not `program.errors.size > 0`: the registry always carries the engine's own built-in + // `HttpError`, whether or not this project's utilities can ever raise it, so `errors.rs` always + // exists (harmlessly — an unused `pub enum` is not `dead_code`) but `error_name`, a private + // driver function, would be if nothing here actually dispatches a fallible entry. + const hasErrors = entries.some((entry) => entry.fails.length > 0); + const driverLines = [ + "// Code generated by the logic engine. DO NOT EDIT.", + "// source: _driver", + "", + "#![allow(clippy::needless_return, unused_parens, unused_imports)]", + "", + ...(usesEnv ? ["use coreout::support::Capabilities;"] : []), + "use coreout::*;", + "", + "use coreout::json;", + "", + ...(hasErrors + ? [ + "// The variant name alone, the same family-membership test `errorName` gives the Go", + "// driver: `CoreError`'s derived `Debug` prints \"VariantName { message: ... }\", so the", + "// first token is it.", + "fn error_name(err: &coreout::errors::CoreError) -> String {", + '\tformat!("{:?}", err).split_whitespace().next().unwrap_or("").to_string()', + "}", + "", + ] + : []), + ...(usesEnv + ? [ + "/// The reference PCG32: same constants and default seed as the interpreter's, so a draw", + "/// matches the reference bit for bit. A fresh instance is built per case, the same way", + "/// the reference model starts a fresh interpreter -- and so a fresh generator -- per case.", + "struct Pcg32 {", + "\tstate: u64,", + "\tincrement: u64,", + "}", + "", + "impl Pcg32 {", + "\tfn new(seed: u64) -> Self {", + "\t\tlet mut p = Pcg32 { state: 0, increment: 1442695040888963407 };", + "\t\tp.next();", + "\t\tp.state = p.state.wrapping_add(seed);", + "\t\tp.next();", + "\t\tp", + "\t}", + "", + "\tfn next(&mut self) -> i64 {", + "\t\tlet previous = self.state;", + "\t\tself.state = previous.wrapping_mul(6364136223846793005).wrapping_add(self.increment);", + "\t\tlet xorshifted = (((previous >> 18) ^ previous) >> 27) as u32;", + "\t\tlet rotation = (previous >> 59) as u32;", + "\t\t(xorshifted.rotate_right(rotation)) as i64", + "\t}", + "}", + "", + "/// The interpreter's own default seed, used whenever Capabilities.seed is left unset.", + "const DEFAULT_SEED: u64 = 0x853c49e6748fea9b;", + "", + "struct Fixture {", + "\tstatus: i64,", + "\tbody: String,", + "\tlatency_millis: i64,", + "}", + "", + "/// The capability fake the differential harness drives: responses come from", + "/// fixtures.json, a missing URL is a transport error, and the scripted latency is what", + "/// decides a race.", + "struct FakeCapabilities {", + "\tfixtures: std::collections::HashMap,", + "\trandom: std::sync::Mutex,", + "}", + "", + "impl Capabilities for FakeCapabilities {", + "\tfn request(&self, request: coreout::support::HttpRequest) -> Option {", + "\t\tlet fixture = self.fixtures.get(&request.url)?;", + "\t\tstd::thread::sleep(std::time::Duration::from_millis(fixture.latency_millis as u64));", + "\t\tSome(coreout::support::HttpResponse { status: fixture.status, headers: vec![], body: fixture.body.clone() })", + "\t}", + "", + "\tfn now(&self) -> i64 { 0 }", + "", + "\tfn sleep(&self, millis: i64) {", + "\t\tstd::thread::sleep(std::time::Duration::from_millis(millis as u64));", + "\t}", + "", + "\tfn next_u32(&self) -> i64 {", + "\t\tself.random.lock().unwrap().next()", + "\t}", + "}", + "", + "fn load_fixtures() -> std::collections::HashMap {", + "\tlet mut fixtures = std::collections::HashMap::new();", + "\tif let Ok(raw) = std::fs::read_to_string(\"fixtures.json\") {", + "\t\tif let json::Json::Object(entries) = json::parse(&raw) {", + "\t\t\tfor (url, value) in entries {", + "\t\t\t\tfixtures.insert(", + "\t\t\t\t\turl,", + "\t\t\t\t\tFixture {", + "\t\t\t\t\t\tstatus: value.get(\"status\").map(|v| v.as_i64()).unwrap_or(0),", + "\t\t\t\t\t\tbody: value.get(\"body\").map(|v| v.as_str().to_string()).unwrap_or_default(),", + "\t\t\t\t\t\tlatency_millis: value.get(\"latencyMillis\").map(|v| v.as_i64()).unwrap_or(0),", + "\t\t\t\t\t},", + "\t\t\t\t);", + "\t\t\t}", + "\t\t}", + "\t}", + "\tfixtures", + "}", + "", + "fn new_environment() -> Box {", + "\tBox::new(FakeCapabilities { fixtures: load_fixtures(), random: std::sync::Mutex::new(Pcg32::new(DEFAULT_SEED)) })", + "}", + "", + ] + : []), + "fn dispatch(name: &str, args: &[json::Json]" + (usesEnv ? ", environment: &dyn Capabilities" : "") + ") -> json::Json {", + "\tmatch name {", + ...cases, + '\t\t_ => coreout::json::Json::Object(vec![("ok".to_string(), coreout::json::Json::Bool(false)), ("error".to_string(), coreout::json::Json::String(format!("unknown function {}", name)))]),', + "\t}", + "}", + "", + "fn main() {", + "\tuse std::io::BufRead;", + "\tlet stdin = std::io::stdin();", + "\tfor line in stdin.lock().lines() {", + "\t\tlet line = line.unwrap();", + "\t\tif line.trim().is_empty() {", + "\t\t\tcontinue;", + "\t\t}", + "\t\tlet parsed = json::parse(&line);", + '\t\tlet name = parsed.get("fn").map(|v| v.as_str().to_string()).unwrap_or_default();', + '\t\tlet args: Vec = parsed.get("args").map(|v| v.as_array().to_vec()).unwrap_or_default();', + ...(usesEnv ? ["\t\tlet environment = new_environment();", "\t\tlet result = dispatch(&name, &args, environment.as_ref());"] : ["\t\tlet result = dispatch(&name, &args);"]), + "\t\tprintln!(\"{}\", json::write(&result));", + "\t}", + "}", + "", + ]; + + return [ + { path: "src/lib.rs", text: libLines.join("\n") }, + // Lives in the library crate, alongside `support`, rather than next to `driver.rs`: Cargo + // auto-discovers every file directly under `src/bin/` as its own binary target (expecting a + // `main` of its own), so a submodule there has to sit in a subdirectory or, more simply + // here, just be a module of `coreout` itself that the driver binary imports. + { path: "src/json.rs", text: JSON_SUPPORT.trim() + "\n" }, + { path: "src/bin/driver.rs", text: driverLines.join("\n") }, + { + path: "Cargo.toml", + text: [ + "[package]", + 'name = "coreout"', + 'version = "0.1.0"', + 'edition = "2021"', + "", + "[[bin]]", + 'name = "driver"', + 'path = "src/bin/driver.rs"', + "", + "[dependencies]", + "", + ].join("\n"), + }, + ]; +} + +/* ------------------------------------------------------------------ * + * printModule needs to record which modules it actually emitted, for `driverFiles` to build + * `lib.rs`'s `pub mod` list from -- `entries` alone only covers modules with an exported utility, + * and an internal-only module (a `lib/*` helper file) would otherwise be missing a `mod` + * declaration despite having a file on disk. + * ------------------------------------------------------------------ */ + +const originalPrintModule = printModule; + +function printModuleTracked(module: TModule): string { + emittedModules.push({ path: module.path }); + return originalPrintModule(module); +} + +export const RUST_BACKEND: Backend = { + spec: RUST_SPEC, + fileExtension: RUST_CONFIG.fileExtension, + // The whole-program pre-pass (see `analysis/borrows.ts` and `docs/decisions/0010-*.md`): the + // shared lowerer only consults this map (`TParam.borrowed`, a "call" node's `borrowedArgs`), it + // never computes it, and it is Rust's own free function precisely so no other target's build + // pays for it or is affected by it. + extraLowerOptions: (program) => ({ borrows: computeBorrowableParams(program) }), + printModule: printModuleTracked, + importPath, + support: supportModule, + errorsModule, + renderType: rustType, + comment: "//", + driver: driverFiles, + // Deliberately absent, unlike Python's and TypeScript's: this engine's own call-site inlining + // (`optimize/inline.ts`) binds every inlined parameter as an owned local — it has no notion of + // a Rust borrow, which is decided later, per call site, by the whole-program pre-pass above. A + // measured attempt at a budget here (`engine/docs/progress.md` §8) inserted a `.to_owned()` at + // every inlined call, one full string clone per splice — exactly the allocation ADR 0010 exists + // to avoid, and by more than the call overhead this pass would have removed. Rust's own + // inliner already reaches the shapes that would matter (`cargo build --release`); teaching this + // pass to reconstruct ADR 0010's borrow analysis for every inlined splice is future work, not + // this one. +}; diff --git a/engine/src/targets/typescript/index.ts b/engine/src/targets/typescript/index.ts new file mode 100644 index 000000000..a11b4f8bb --- /dev/null +++ b/engine/src/targets/typescript/index.ts @@ -0,0 +1,1095 @@ +/** + * The TypeScript backend. + * + * ES2020 on Node 20, ESM only, one module per source module with named exports, no module level + * effects and no external dependencies. Integers are `number` while their proven range fits + * ±(2^53−1) and `bigint` otherwise; a `Decimal` is its unscaled integer, which is exact and + * native everywhere. + */ + +import type { SemType } from "../../types.ts"; +import { SAFE_INT_HI, SAFE_INT_LO, typeToString } from "../../types.ts"; +import type { Value } from "../../values.ts"; +import type { Candidate } from "../../backend/select.ts"; +import { LoweringTable, argIsAscii } from "../../backend/select.ts"; +import type { TargetSpec } from "../../backend/lower.ts"; +import { asciiString, mapExprs } from "../../backend/tast.ts"; +import type { TExpr, TFunc, TModule, TRecord, TStmt } from "../../backend/tast.ts"; + +export const TYPESCRIPT_CONFIG = { + baseline: "ES2020 on Node 20", + fileExtension: ".ts", + /** Node runs the generated sources directly, so relative imports carry their extension. */ + importExtension: ".ts", + dependencies: [] as string[], + formatter: "prettier", + linters: ["tsc --strict --noEmit"], +}; + +/* ------------------------------------------------------------------ * + * Types + * ------------------------------------------------------------------ */ + +export function needsBigInt(type: SemType): boolean { + return type.kind === "Int" && (type.lo < SAFE_INT_LO || type.hi > SAFE_INT_HI); +} + +export function tsType(type: SemType): string { + switch (type.kind) { + case "Bool": + return "boolean"; + case "Int": + return needsBigInt(type) ? "bigint" : "number"; + case "Float": + return "number"; + case "Decimal": + return "number"; + case "String": + return "string"; + case "List": + return `readonly ${wrap(tsType(type.elem))}[]`; + case "Option": + return `${tsType(type.inner)} | undefined`; + case "Record": + return type.name; + case "Enum": + return type.members.map((member) => JSON.stringify(member)).join(" | "); + case "Union": + return type.name; + case "CivilDate": + case "Instant": + case "Duration": + return "number"; + case "Lambda": + return `(${type.params.map((param, index) => `arg${index}: ${tsType(param)}`).join(", ")}) => ${tsType(type.ret)}`; + case "Void": + return "void"; + case "Never": + return "never"; + default: { + const exhaustive: never = type; + return exhaustive; + } + } +} + +/** The type of a local that is still being built: a list is mutable until it escapes. */ +export function mutableType(type: SemType): string { + return type.kind === "List" ? `${wrap(tsType(type.elem))}[]` : tsType(type); +} + +function wrap(rendered: string): string { + return rendered.includes("|") || rendered.includes("=>") ? `(${rendered})` : rendered; +} + +/* ------------------------------------------------------------------ * + * Capability table + * ------------------------------------------------------------------ */ + + +/** The body of a printed class, so a lowering can negate it. */ +function classBody(printed: string): string { + return printed.startsWith("[") && printed.endsWith("]") ? printed.slice(1, -1) : printed; +} + +const raw = (text: string): TExpr => ({ kind: "raw", text }); + +function binary(op: string): Candidate["emit"] { + return (args) => ({ kind: "binary", op, left: args[0]!, right: args[1]! }); +} + +function method(name: string, extra = 0): Candidate["emit"] { + return (args) => ({ kind: "method", target: args[0]!, name, args: args.slice(1, 1 + extra) }); +} + +const cheap = { alloc: "none", time: "constant" } as const; +const linear = { alloc: "none", time: "linear" } as const; +const allocating = { alloc: "one", time: "linear" } as const; +/** A pass that materializes the scalars of a string: correct everywhere, and the slowest option. */ +const scalarPass = { alloc: "many", time: "linear" } as const; + +export const TYPESCRIPT_CANDIDATES: readonly Candidate[] = [ + // Integers and floats + ...["add:+", "sub:-", "mul:*"].map((entry) => { + const [op, symbol] = entry.split(":") as [string, string]; + return { op: `int.${op}`, impl: "native" as const, cost: cheap, emit: binary(symbol) }; + }), + ...["add:+", "sub:-", "mul:*", "div:/"].map((entry) => { + const [op, symbol] = entry.split(":") as [string, string]; + return { op: `float.${op}`, impl: "native" as const, cost: cheap, emit: binary(symbol) }; + }), + { + op: "int.div", + impl: "native", + cost: cheap, + because: "`/` on numbers is not integer division", + emit: (args, types) => + needsBigInt(types[0] ?? { kind: "Int", lo: 0n, hi: 0n }) + ? { kind: "binary", op: "/", left: args[0]!, right: args[1]! } + : { kind: "call", callee: raw("Math.trunc"), args: [{ kind: "binary", op: "/", left: args[0]!, right: args[1]! }] }, + }, + { op: "int.mod", impl: "native", cost: cheap, emit: binary("%") }, + { op: "int.neg", impl: "native", cost: cheap, emit: (args) => ({ kind: "unary", op: "-", operand: args[0]! }) }, + { op: "float.neg", impl: "native", cost: cheap, emit: (args) => ({ kind: "unary", op: "-", operand: args[0]! }) }, + { op: "int.abs", impl: "native", cost: cheap, emit: (args) => ({ kind: "call", callee: raw("Math.abs"), args: [args[0]!] }) }, + { op: "int.min", impl: "native", cost: cheap, emit: (args) => ({ kind: "call", callee: raw("Math.min"), args }) }, + { op: "int.max", impl: "native", cost: cheap, emit: (args) => ({ kind: "call", callee: raw("Math.max"), args }) }, + ...["lt:<", "le:<=", "gt:>", "ge:>="].flatMap((entry) => { + const [op, symbol] = entry.split(":") as [string, string]; + return [ + { op: `int.${op}`, impl: "native" as const, cost: cheap, emit: binary(symbol) }, + { op: `float.${op}`, impl: "native" as const, cost: cheap, emit: binary(symbol) }, + ]; + }), + { op: "float.fromInt", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { op: "core.eq", impl: "native", cost: cheap, emit: binary("===") }, + + // Options + { op: "opt.isNone", impl: "native", cost: cheap, emit: (args) => ({ kind: "binary", op: "===", left: args[0]!, right: raw("undefined") }) }, + { op: "opt.unwrap", impl: "native", cost: cheap, emit: (args) => ({ kind: "unary", op: "post!", operand: args[0]! }) }, + { op: "opt.some", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { op: "opt.orElse", impl: "native", cost: cheap, emit: binary("??") }, + + // Strings + { + op: "str.len", + impl: "native", + requires: argIsAscii(0), + because: "`String#length` counts UTF-16 code units, which only equals the scalar count for ASCII", + cost: cheap, + emit: (args) => ({ kind: "member", target: args[0]!, name: "length" }), + }, + { + op: "str.len", + impl: "portable", + cost: allocating, + emit: (args) => ({ kind: "member", target: { kind: "raw", text: `[...${print(args[0]!)}]` }, name: "length" }), + }, + { op: "str.concat", impl: "native", cost: cheap, emit: binary("+") }, + { + op: "str.codeAt", + impl: "native", + requires: argIsAscii(0), + because: "`charCodeAt` returns a UTF-16 code unit", + cost: cheap, + emit: method("charCodeAt", 1), + }, + { + op: "str.charAt", + impl: "native", + requires: argIsAscii(0), + because: "indexing a string yields one UTF-16 code unit", + cost: cheap, + emit: (args) => ({ kind: "index", target: args[0]!, index: args[1]! }), + }, + { + op: "str.codeAtOpt", + impl: "native", + requires: argIsAscii(0), + because: "charCodeAt answers NaN past the end, so the bound is checked explicitly", + cost: cheap, + emit: (args) => + raw( + `(${print(args[1]!)} < ${print(args[0]!)}.length ? ${print(args[0]!)}.charCodeAt(${print(args[1]!)}) : undefined)`, + ), + }, + { + op: "str.charAtOpt", + impl: "native", + requires: argIsAscii(0), + because: "indexing a string yields one UTF-16 code unit, and undefined past the end", + cost: cheap, + emit: (args) => raw(`${print(args[0]!)}[${print(args[1]!)}]`), + }, + { + op: "str.slice", + impl: "native", + requires: argIsAscii(0), + because: "`slice` cuts at UTF-16 boundaries", + cost: allocating, + emit: method("slice", 2), + }, + { op: "str.indexOf", impl: "native", requires: argIsAscii(0), because: "`indexOf` returns a UTF-16 index", cost: linear, emit: method("indexOf", 1) }, + { op: "str.contains", impl: "native", cost: linear, emit: method("includes", 1) }, + { op: "str.startsWith", impl: "native", cost: linear, emit: method("startsWith", 1) }, + { op: "str.endsWith", impl: "native", cost: linear, emit: method("endsWith", 1) }, + { op: "str.repeat", impl: "native", cost: allocating, emit: method("repeat", 1) }, + { op: "str.padStart", impl: "native", requires: argIsAscii(0), because: "`padStart` counts UTF-16 code units", cost: allocating, emit: method("padStart", 2) }, + { + op: "str.trim", + impl: "native", + cost: allocating, + because: "`String#trim` removes exactly the 25 code points the spec names", + emit: method("trim"), + }, + { + op: "str.asciiUpper", + impl: "native", + requires: argIsAscii(0), + because: "`toUpperCase` is only ASCII-equivalent on ASCII input (ß becomes SS otherwise)", + cost: allocating, + emit: method("toUpperCase"), + }, + { + op: "str.asciiLower", + impl: "native", + requires: argIsAscii(0), + because: "`toLowerCase` is only ASCII-equivalent on ASCII input", + cost: allocating, + emit: method("toLowerCase"), + }, + { + op: "str.asciiUpper", + impl: "native", + because: "a single regex pass maps a-z and leaves every other scalar alone, which is the Core's rule for any input", + cost: allocating, + emit: (args) => raw(`${print(args[0]!)}.replace(/[a-z]/gu, (scalar) => scalar.toUpperCase())`), + }, + { + op: "str.asciiLower", + impl: "native", + because: "a single regex pass maps A-Z and leaves every other scalar alone", + cost: allocating, + emit: (args) => raw(`${print(args[0]!)}.replace(/[A-Z]/gu, (scalar) => scalar.toLowerCase())`), + }, + { + op: "str.compare", + impl: "native", + requires: (args) => argIsAscii(0)(args) && argIsAscii(1)(args), + because: "`<` compares UTF-16 code units, so astral scalars would sort before U+E000", + cost: cheap, + emit: (args) => raw(`(${print(args[0]!)} < ${print(args[1]!)} ? -1 : ${print(args[0]!)} > ${print(args[1]!)} ? 1 : 0)`), + }, + { + op: "str.compare", + impl: "portable", + cost: scalarPass, + sourceFn: "std/strings::compareScalars", + emit: (args, _types, ctx) => ({ + kind: "call", + callee: { kind: "name", name: ctx.nameOf("std/strings::compareScalars") }, + args: [args[0]!, args[1]!], + }), + }, + { + op: "str.codePoints", + impl: "native", + cost: allocating, + emit: (args) => raw(`Array.from(${print(args[0]!)}, (scalar) => scalar.codePointAt(0)!)`), + }, + { + op: "str.fromCodePoints", + impl: "native", + cost: allocating, + emit: (args) => raw(`${print(args[0]!)}.map((point) => String.fromCodePoint(point)).join("")`), + }, + { + op: "str.asAscii", + impl: "native", + cost: linear, + emit: (args) => raw(`(Array.from(${print(args[0]!)}).every((scalar) => scalar.codePointAt(0)! < 0x80) ? ${print(args[0]!)} : undefined)`), + }, + { + op: "str.asDigits", + impl: "native", + cost: linear, + emit: (args) => raw(`(/^[0-9]+$/.test(${print(args[0]!)}) ? ${print(args[0]!)} : undefined)`), + }, + { op: "str.split", impl: "native", requires: argIsAscii(1), because: "the separator must be one ASCII scalar", cost: allocating, emit: method("split", 1) }, + { op: "str.join", impl: "native", cost: allocating, emit: method("join", 1) }, + { op: "str.fromInt", impl: "native", cost: allocating, emit: (args) => ({ kind: "method", target: args[0]!, name: "toString", args: [] }) }, + { + op: "str.parseInt", + impl: "native", + cost: linear, + emit: (args) => raw(`(/^[0-9]{1,18}$/.test(${print(args[0]!)}) ? Number(${print(args[0]!)}) : undefined)`), + }, + + // Sequences + { + op: "seq.at", + impl: "native", + cost: cheap, + emit: (args) => raw(`${print(args[0]!)}[${print(args[1]!)}]`), + }, + { + op: "str.asciiUpper", + impl: "portable", + // Building a scalar list costs far more than the host's own pass, which is why the cost + // class has to say so: selection ranks by cost before it ranks by implementation kind. + cost: scalarPass, + sourceFn: "std/strings::asciiUpperAll", + emit: (args, _types, ctx) => ({ kind: "call", callee: { kind: "name", name: ctx.nameOf("std/strings::asciiUpperAll") }, args: [args[0]!] }), + }, + { + op: "str.asciiLower", + impl: "portable", + cost: scalarPass, + sourceFn: "std/strings::asciiLowerAll", + emit: (args, _types, ctx) => ({ kind: "call", callee: { kind: "name", name: ctx.nameOf("std/strings::asciiLowerAll") }, args: [args[0]!] }), + }, + { op: "seq.len", impl: "native", cost: cheap, emit: (args) => ({ kind: "member", target: args[0]!, name: "length" }) }, + { op: "seq.get", impl: "native", cost: cheap, emit: (args) => ({ kind: "index", target: args[0]!, index: args[1]! }) }, + { op: "seq.push", impl: "native", cost: cheap, emit: (args) => ({ kind: "method", target: args[0]!, name: "push", args: [args[1]!] }) }, + { op: "seq.map", impl: "native", cost: allocating, emit: method("map", 1) }, + { op: "seq.filter", impl: "native", cost: allocating, emit: method("filter", 1) }, + { op: "seq.any", impl: "native", cost: linear, emit: method("some", 1) }, + { op: "seq.all", impl: "native", cost: linear, emit: method("every", 1) }, + { op: "seq.find", impl: "native", cost: linear, emit: method("find", 1) }, + { op: "seq.fold", impl: "native", cost: linear, emit: (args) => ({ kind: "method", target: args[0]!, name: "reduce", args: [args[2]!, args[1]!] }) }, + { op: "seq.sum", impl: "native", cost: linear, emit: (args) => raw(`${print(args[0]!)}.reduce((total, item) => total + item, 0)`) }, + { op: "seq.indexOf", impl: "native", cost: linear, emit: method("indexOf", 1) }, + { op: "seq.contains", impl: "native", cost: linear, emit: method("includes", 1) }, + { op: "seq.concat", impl: "native", cost: allocating, emit: method("concat", 1) }, + { op: "seq.slice", impl: "native", cost: allocating, emit: method("slice", 2) }, + { op: "seq.reverse", impl: "native", cost: allocating, emit: (args) => raw(`[...${print(args[0]!)}].reverse()`) }, + { + op: "seq.sortStable", + impl: "native", + because: "Array#sort has been required to be stable since ES2019", + cost: { alloc: "one", time: "nlogn" }, + emit: (args) => raw(`[...${print(args[0]!)}].sort(${print(args[1]!)})`), + }, + { + op: "seq.sortStableBy", + impl: "native", + cost: { alloc: "one", time: "nlogn" }, + emit: (args, types) => { + const key = print(args[1]!); + const keyType = types[1]; + const comparison = + keyType !== undefined && keyType.kind === "Lambda" && keyType.ret.kind === "String" + ? "left < right ? -1 : left > right ? 1 : 0" + : "left < right ? -1 : left > right ? 1 : 0"; + return raw( + `[...${print(args[0]!)}].sort((a, b) => { const left = (${key})(a); const right = (${key})(b); return ${comparison}; })`, + ); + }, + }, + + // Decimals: the unscaled integer, which is exact and native. + { op: "dec.fromScaled", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { op: "dec.fromInt", impl: "native", cost: cheap, emit: (args, types) => raw(`${print(args[0]!)} * ${10 ** scaleOf(types[1])}`) }, + { op: "dec.add", impl: "native", cost: cheap, emit: binary("+") }, + { op: "dec.sub", impl: "native", cost: cheap, emit: binary("-") }, + { op: "dec.mul", impl: "native", cost: cheap, emit: binary("*") }, + { op: "dec.compare", impl: "native", cost: cheap, emit: (args) => raw(`(${print(args[0]!)} < ${print(args[1]!)} ? -1 : ${print(args[0]!)} > ${print(args[1]!)} ? 1 : 0)`) }, + { op: "dec.isNegative", impl: "native", cost: cheap, emit: (args) => raw(`${print(args[0]!)} < 0`) }, + { op: "dec.abs", impl: "native", cost: cheap, emit: (args) => ({ kind: "call", callee: raw("Math.abs"), args: [args[0]!] }) }, + { op: "dec.unscaled", impl: "native", cost: cheap, emit: (args) => args[0]! }, + + // Dates: a civil date is days since the epoch in every target. + { + op: "date.clampEpochDays", + impl: "native", + cost: cheap, + emit: (args) => raw(`Math.min(Math.max(${print(args[0]!)}, -719162), 2932896)`), + }, + { op: "date.toEpochDays", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { + op: "date.fromEpochDays", + impl: "native", + cost: cheap, + emit: (args) => raw(`(${print(args[0]!)} >= -719162 && ${print(args[0]!)} <= 2932896 ? ${print(args[0]!)} : undefined)`), + }, + { + // `ymdToDays` (the fallback below) computes the day forward and then verifies the round trip + // by decomposing it back into year/month/day through three more floor-division-heavy Hinnant + // functions — measured at 78% of `getHolidays`' call (`engine/docs/progress.md` §8). The round + // trip only exists to answer one question, "is `day` within the month it names", which is + // exactly what a days-in-month table already answers directly: once the month and year are in + // range, `day` names a real date iff it does not exceed that month's length (28-31, with + // February's leap adjustment). That table check plus the single forward computation is + // mathematically the same predicate the round trip computes, just without decomposing the + // result back out again. + op: "date.fromYmd", + impl: "portable", + cost: cheap, + sourceFn: "std/date::daysFromCivil", + emit: (args, _types, ctx) => { + // A month the caller has already settled is the common case once a helper is inlined into + // a call site that passes one (`easterSunday` builds a March date), and it decides both + // tests below: the range check is then a fact, not code, and the days-in-month ladder + // collapses to one number — except in February, where the year still decides. This is the + // same folding `backend/fold.ts` does for the target AST, done here because this lowering + // emits text, which that pass cannot see into. + const literalMonth = ((): number | undefined => { + const month = args[1]!; + if (month.kind !== "lit") return undefined; + if (typeof month.value === "bigint") return Number(month.value); + return typeof month.value === "number" && Number.isInteger(month.value) ? month.value : undefined; + })(); + const monthOutOfRange = literalMonth !== undefined && (literalMonth < 1 || literalMonth > 12); + const monthTest = (month: string): string => (literalMonth === undefined ? `${month} < 1 || ${month} > 12 || ` : ""); + const dayLimit = (year: string, month: string): string => { + if (literalMonth === 2) return `((${year} % 4 === 0 && ${year} % 100 !== 0) || ${year} % 400 === 0 ? 29 : 28)`; + if (literalMonth !== undefined) return String([4, 6, 9, 11].includes(literalMonth) ? 30 : 31); + return `${month} === 2 ? ((${year} % 4 === 0 && ${year} % 100 !== 0) || ${year} % 400 === 0 ? 29 : 28) : (${month} === 4 || ${month} === 6 || ${month} === 9 || ${month} === 11 ? 30 : 31)`; + }; + const forward = (year: string, month: string, day: string): TExpr => + ({ + kind: "call", + callee: { kind: "name", name: ctx.nameOf("std/date::daysFromCivil") }, + args: [raw(year), raw(month), raw(day)], + }) as TExpr; + // A cheap-to-duplicate argument (a name or a literal, never a call or a computed member) is + // printed directly, several times, rather than paying for a closure that only exists to + // bind it once — the same reasoning `date.fromEpochDays` above already relies on. Anything + // else is bound once by an IIFE, so an effectful or otherwise non-trivial expression is + // never evaluated twice. + if (args.every((arg) => arg.kind === "name" || arg.kind === "lit")) { + const year = print(args[0]!); + const month = print(args[1]!); + const day = print(args[2]!); + if (monthOutOfRange) return raw("undefined"); + return raw( + `(${year} < 1 || ${year} > 9999 || ${monthTest(month)}${day} < 1 || ${day} > (${dayLimit(year, month)}) ? undefined : ${print(forward(year, month, day))})`, + ); + } + if (monthOutOfRange) { + return raw(`((y, m, d) => undefined)(${print(args[0]!)}, ${print(args[1]!)}, ${print(args[2]!)})`); + } + return raw( + `((y, m, d) => {\n` + + `\t\tif (y < 1 || y > 9999 || ${monthTest("m")}d < 1) return undefined;\n` + + `\t\tconst limit = ${dayLimit("y", "m")};\n` + + `\t\treturn d > limit ? undefined : ${print(forward("y", "m", "d"))};\n` + + `\t})(${print(args[0]!)}, ${print(args[1]!)}, ${print(args[2]!)})`, + ); + }, + }, + { + op: "date.year", + impl: "portable", + cost: cheap, + sourceFn: "std/date::yearFromDays", + emit: (args, _types, ctx) => ({ kind: "call", callee: { kind: "name", name: ctx.nameOf("std/date::yearFromDays") }, args: [args[0]!] }), + }, + { + op: "date.month", + impl: "portable", + cost: cheap, + sourceFn: "std/date::monthFromDays", + emit: (args, _types, ctx) => ({ kind: "call", callee: { kind: "name", name: ctx.nameOf("std/date::monthFromDays") }, args: [args[0]!] }), + }, + { + op: "date.day", + impl: "portable", + cost: cheap, + sourceFn: "std/date::dayFromDays", + emit: (args, _types, ctx) => ({ kind: "call", callee: { kind: "name", name: ctx.nameOf("std/date::dayFromDays") }, args: [args[0]!] }), + }, + { + op: "date.addDays", + impl: "native", + cost: cheap, + emit: (args) => + raw( + `(${print(args[0]!)} + ${print(args[1]!)} >= -719162 && ${print(args[0]!)} + ${print(args[1]!)} <= 2932896 ? ${print(args[0]!)} + ${print(args[1]!)} : undefined)`, + ), + }, + { op: "date.diffDays", impl: "native", cost: cheap, emit: binary("-") }, + { op: "date.compare", impl: "native", cost: cheap, emit: (args) => raw(`(${print(args[0]!)} < ${print(args[1]!)} ? -1 : ${print(args[0]!)} > ${print(args[1]!)} ? 1 : 0)`) }, + { op: "date.dayOfWeek", impl: "native", cost: cheap, emit: (args) => raw(`((((${print(args[0]!)} + 3) % 7) + 7) % 7) + 1`) }, + { + op: "date.isLeapYear", + impl: "native", + cost: cheap, + emit: (args) => raw(`((${print(args[0]!)} % 4 === 0 && ${print(args[0]!)} % 100 !== 0) || ${print(args[0]!)} % 400 === 0)`), + }, + + // Regex: the normalized pattern, printed in the dialect every target reads the same way. + { + op: "re.retain", + impl: "native", + because: "a single pass with the negated class, which every engine reads the same way", + cost: allocating, + emit: (args, _types, ctx) => { + const pattern = ctx.regex === undefined ? "" : printRegexNode(ctx.regex.node, "javascript"); + return raw(`${print(args[0]!)}.replace(/[^${classBody(pattern)}]/gu, "")`); + }, + }, + { + op: "re.test", + impl: "native", + cost: linear, + because: "the normalized pattern is inside the compatibility subset", + emit: (args, _types, ctx) => raw(`/^${printedPattern(ctx)}$/u.test(${print(args[0]!)})`), + }, + + // Capabilities + { + op: "http.request", + impl: "native", + cost: { alloc: "many", time: "linear" }, + emit: (args, _types, ctx) => ({ kind: "method", target: ctx.env(), name: "request", args: [args[0]!], await: true }), + }, + { op: "clock.now", impl: "native", cost: cheap, emit: (_args, _types, ctx) => ({ kind: "method", target: ctx.env(), name: "now", args: [] }) }, + { + op: "clock.sleep", + impl: "native", + cost: cheap, + emit: (args, _types, ctx) => ({ kind: "method", target: ctx.env(), name: "sleep", args: [args[0]!], await: true }), + }, + { op: "clock.millis", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { op: "clock.durationMillis", impl: "native", cost: cheap, emit: (args) => args[0]! }, + { op: "clock.elapsed", impl: "native", cost: cheap, emit: (args) => raw(`Math.max(0, ${print(args[1]!)} - ${print(args[0]!)})`) }, + { op: "random.nextU32", impl: "native", cost: cheap, emit: (_args, _types, ctx) => ({ kind: "method", target: ctx.env(), name: "nextU32", args: [] }) }, + { + op: "task.race", + impl: "native", + because: "Promise.any resolves with the first task to answer, which is the semantics of race", + cost: { alloc: "many", time: "linear" }, + emit: (args) => raw(`await raceFirstSome(${print(args[0]!)})`), + }, +]; + +function scaleOf(type: SemType | undefined): number { + return type !== undefined && type.kind === "Int" ? Number(type.lo) : 0; +} + +function printedPattern(ctx: { regex?: { node: unknown } }): string { + const regex = ctx.regex as { node: Parameters[0] } | undefined; + return regex === undefined ? "" : printRegexNode(regex.node, "javascript"); +} + +import { printRegex as printRegexNode } from "../../regex.ts"; + +export const TYPESCRIPT_SPEC: TargetSpec = { + name: "typescript", + table: new LoweringTable(TYPESCRIPT_CANDIDATES), + naming: { + func: (name) => camel(name), + value: (name) => camel(name), + field: (name) => name, + type: (name) => name, + module: (path) => `${path}.ts`, + }, + loopCombinators: new Set(), + statementTernary: false, + errorsAsValues: false, + asyncColouring: true, + envType: { kind: "Record", name: "Capabilities" }, +}; + +function camel(name: string): string { + return name.replace(/[-_](.)/g, (_match, char: string) => char.toUpperCase()); +} + +/* ------------------------------------------------------------------ * + * Printer + * ------------------------------------------------------------------ */ + +export function print(expr: TExpr): string { + switch (expr.kind) { + case "lit": + return literal(expr.value, expr.type); + case "name": + return expr.name; + case "raw": + return expr.text; + case "call": + return `${expr.await === true ? "await " : ""}${print(expr.callee)}(${expr.args.map(print).join(", ")})`; + case "method": + return `${expr.await === true ? "await " : ""}${print(expr.target)}.${expr.name}(${expr.args.map(print).join(", ")})`; + case "member": + return `${print(expr.target)}.${expr.name}`; + case "index": + return `${print(expr.target)}[${print(expr.index)}]`; + case "binary": + return `(${print(expr.left)} ${expr.op} ${print(expr.right)})`; + case "unary": { + if (expr.op === "post!") return `${print(expr.operand)}!`; + // `!(a === b)` reads better as `a !== b`, and the formatter cannot do that for us. + if (expr.op === "!" && expr.operand.kind === "binary" && expr.operand.op === "===") { + return `(${print(expr.operand.left)} !== ${print(expr.operand.right)})`; + } + return `${expr.op}${print(expr.operand)}`; + } + case "ternary": + return `(${print(expr.test)} ? ${print(expr.then)} : ${print(expr.otherwise)})`; + case "list": + return `[${expr.items.map(print).join(", ")}]`; + case "record": + return `{ ${expr.fields.map((field) => `${field.name}: ${print(field.value)}`).join(", ")} }`; + case "lambda": { + const params = expr.params.map((param) => `${param.name}: ${tsType(param.type)}`).join(", "); + // A lambda that awaits is async, which is the same colouring rule the compiler applies + // to a function that reaches Http. + const asyncPrefix = containsAwait(expr.body) ? "async " : ""; + const ret = asyncPrefix === "" ? tsType(expr.ret) : `Promise<${tsType(expr.ret)}>`; + if (expr.body.length === 1 && expr.body[0]!.kind === "return" && expr.body[0]!.value !== undefined) { + return `${asyncPrefix}(${params}): ${ret} => ${print(expr.body[0]!.value!)}`; + } + return `${asyncPrefix}(${params}): ${ret} => {\n${printBody(expr.body, 1)}\n}`; + } + case "none": + return "undefined"; + case "some": + return print(expr.inner); + case "zero": + return "undefined"; + default: { + const exhaustive: never = expr; + return exhaustive; + } + } +} + +/** Whether a statement list awaits anywhere outside a nested lambda. */ +function containsAwait(body: readonly TStmt[]): boolean { + let found = false; + mapExprs(body, (expr) => { + if ((expr.kind === "call" || expr.kind === "method") && expr.await === true) found = true; + return expr; + }); + return found; +} + +function literal(value: Value, type: SemType): string { + if (typeof value === "bigint") return needsBigInt(type) ? `${value}n` : value.toString(); + if (typeof value === "string") return asciiString(value); + if (typeof value === "boolean") return String(value); + if (typeof value === "number") return String(value); + if (Array.isArray(value)) { + const elem = type.kind === "List" ? type.elem : type; + return `[${value.map((item) => literal(item, elem)).join(", ")}]`; + } + return "undefined"; +} + +function printBody(body: readonly TStmt[], depth: number): string { + return body.map((statement) => printStmt(statement, depth)).join("\n"); +} + +function printStmt(statement: TStmt, depth: number): string { + const pad = "\t".repeat(depth); + switch (statement.kind) { + case "let": { + // A list still being built is mutable, so it is not printed `readonly`; it is frozen by + // the semantics as soon as it escapes, which the Core guarantees. + const rendered = statement.mutable ? mutableType(statement.type) : tsType(statement.type); + return `${pad}${statement.mutable ? "let" : "const"} ${statement.name}: ${rendered} = ${print(statement.init)};`; + } + case "multiLet": + return `${pad}const [${statement.names.join(", ")}] = ${print(statement.init)};`; + case "assign": + return `${pad}${print(statement.target)} = ${print(statement.value)};`; + case "if": { + const head = `${pad}if (${print(statement.test)}) {\n${printBody(statement.then, depth + 1)}\n${pad}}`; + return statement.otherwise.length === 0 + ? head + : `${head} else {\n${printBody(statement.otherwise, depth + 1)}\n${pad}}`; + } + case "switch": { + // The Core's `switch` never falls through — each case is a self-contained branch, which + // is exactly why the source's own case-closing `break` carries no meaning and is dropped + // on the way into Core (see `caseBody` in `core/check.ts`). JavaScript's `switch` is the + // opposite: without an explicit `break`, one matching case runs every case below it too. + // Printing the Core's cases as bare `case`/`default` blocks would silently reintroduce + // the fallthrough the Core specifically does not have, so every case ends with a `break` + // here, regardless of whether its own body already exits (a `break` after a `return` is + // unreachable, not wrong, and is cheaper to emit unconditionally than to prove unneeded). + const closedBody = (body: readonly TStmt[], indent: number): string => + body.length === 0 + ? `${"\t".repeat(indent)}break;` + : `${printBody(body, indent)}\n${"\t".repeat(indent)}break;`; + const cases = statement.cases + .map( + (entry) => + `${pad}\t${entry.values.map((value) => `case ${JSON.stringify(value)}:`).join("\n" + pad + "\t")}\n${closedBody(entry.body, depth + 2)}`, + ) + .join("\n"); + const fallback = + statement.otherwise === undefined + ? "" + : `\n${pad}\tdefault:\n${closedBody(statement.otherwise, depth + 2)}`; + return `${pad}switch (${print(statement.subject)}) {\n${cases}${fallback}\n${pad}}`; + } + case "for": { + const comparison = statement.step > 0n ? (statement.inclusive ? "<=" : "<") : statement.inclusive ? ">=" : ">"; + const update = statement.step === 1n ? `${statement.name}++` : statement.step === -1n ? `${statement.name}--` : `${statement.name} += ${statement.step}`; + return `${pad}for (let ${statement.name} = ${print(statement.from)}; ${statement.name} ${comparison} ${print(statement.to)}; ${update}) {\n${printBody(statement.body, depth + 1)}\n${pad}}`; + } + case "forEach": + return `${pad}for (const ${statement.name} of ${print(statement.iterable)}) {\n${printBody(statement.body, depth + 1)}\n${pad}}`; + case "return": + return statement.value === undefined ? `${pad}return;` : `${pad}return ${print(statement.value)};`; + case "throw": + return `${pad}throw new ${statement.errorClass}(${statement.args.map(print).join(", ")});`; + case "break": + return `${pad}break;`; + case "continue": + return `${pad}continue;`; + case "expr": + return `${pad}${print(statement.expr)};`; + case "raw": + return `${pad}${statement.text}`; + default: { + const exhaustive: never = statement; + return exhaustive; + } + } +} + +export function printFunction(fn: TFunc): string { + const params = fn.params.map((param) => `${param.name}: ${tsType(param.type)}`).join(", "); + const ret = fn.isAsync ? `Promise<${tsType(fn.ret)}>` : tsType(fn.ret); + const doc = fn.doc === undefined ? "" : `/**\n${fn.doc.split("\n").map((line) => ` * ${line}`.trimEnd()).join("\n")}\n */\n`; + // `moduleExported` is the source module's own `export`, not `exported` (utility-ness): a + // helper never declared `export function` in its source stays a plain, unexported function + // here too, even when it is a public utility's own internal implementation detail. + const modifier = fn.moduleExported ? "export " : ""; + return `${doc}${modifier}${fn.isAsync ? "async " : ""}function ${fn.name}(${params}): ${ret} {\n${printBody(fn.body, 1)}\n}`; +} + +export function printRecord(record: TRecord): string { + const doc = record.doc === undefined ? "" : `/** ${record.doc} */\n`; + const fields = record.fields + .map((field) => `${field.doc === undefined ? "" : `\t/** ${field.doc} */\n`}\t readonly ${field.name}: ${tsType(field.type)};`) + .join("\n") + .replaceAll("\t readonly", "\treadonly"); + return `${doc}export type ${record.name} = {\n${fields}\n};`; +} + +export function printModule(module: TModule): string { + const parts: string[] = [module.header]; + void module.requires; + for (const item of module.imports) { + // `capabilities` lives at the output root, so a nested module walks back up to it. + const from = item.from === "SUPPORT" ? importPath(module.sourcePath, "capabilities") : item.from; + parts.push( + `import ${item.typeOnly === true ? "type " : ""}{ ${item.names.join(", ")} } from ${JSON.stringify(from)};`, + ); + } + if (module.imports.length > 0) parts.push(""); + for (const record of module.records) parts.push(printRecord(record), ""); + for (const constant of module.constants) { + parts.push(`const ${constant.name}: ${tsType(constant.type)} = ${print(constant.value)};`, ""); + } + for (const fn of module.functions) parts.push(printFunction(fn), ""); + return `${parts.join("\n").trimEnd()}\n`; +} + +export { typeToString }; + +/* ------------------------------------------------------------------ * + * Backend + * ------------------------------------------------------------------ */ + +import { dirname, relative as relativePath } from "node:path"; +import type { Backend, DriverEntry, SupportNeeds } from "../../backend/generate.ts"; +import type { CProgram } from "../../core/ir.ts"; + +function importPath(from: string, to: string): string { + const rel = relativePath(dirname(from) === "." ? "" : dirname(from), to).split("\\").join("/"); + const prefixed = rel.startsWith(".") ? rel : `./${rel}`; + return `${prefixed}${TYPESCRIPT_CONFIG.importExtension}`; +} + +/** + * The generated capability defaults. + * + * This is generated code, not a runtime package: it uses only the standard library, it is written + * into the output next to the utilities, and a project that fakes its capabilities in tests simply + * passes a different object. + */ +function supportModule(_program: CProgram, needs: SupportNeeds): { path: string; text: string } | undefined { + if (!needs.env) return undefined; + const race = ` +/** + * Takes the first task to answer, discarding the losers, and answers undefined when none does. + * + * Cancellation is best effort and semantically unobservable: a losing task may keep running, and + * its answer is dropped. Only idempotent work belongs inside a race. + */ +export async function raceFirstSome(tasks: readonly (() => Promise)[]): Promise { + try { + return await Promise.any( + tasks.map(async (task) => { + const value = await task(); + + if (value === undefined) throw new Error("no answer"); + + return value; + }), + ); + } catch { + return undefined; + } +} +`; + const text = `// Code generated by the logic engine. DO NOT EDIT. +// engine: ${ENGINE_VERSION} +// source: capabilities + +/** One request or response header. */ +export type HttpHeader = { + readonly name: string; + readonly value: string; +}; + +/** An Http request handed to the environment. */ +export type HttpRequest = { + readonly method: string; + readonly url: string; + readonly headers: readonly HttpHeader[]; + readonly body: string; + readonly timeoutMillis: number; +}; + +/** An Http response. A status of 400 or more is a value, not a failure. */ +export type HttpResponse = { + readonly status: number; + readonly headers: readonly HttpHeader[]; + readonly body: string; +}; + +/** Everything the core needs from the outside world. */ +export type Capabilities = { + /** A transport error or a timeout answers \`undefined\`; a 4xx or 5xx status is a value. */ + request(request: HttpRequest): Promise; + now(): number; + sleep(milliseconds: number): Promise; + nextU32(): number; +}; + +/** The default environment, built from the platform's own standard library. */ +export function defaultCapabilities(): Capabilities { + return { + async request(request: HttpRequest): Promise { + const controller = new AbortController(); + const timer = setTimeout(() => controller.abort(), request.timeoutMillis); + + try { + const response = await fetch(request.url, { + method: request.method, + headers: request.headers.map((header) => [header.name, header.value]), + body: request.body === "" ? undefined : request.body, + signal: controller.signal, + }); + + const headers: { name: string; value: string }[] = []; + response.headers.forEach((value, name) => headers.push({ name, value })); + + return { status: response.status, headers, body: await response.text() }; + } catch { + return undefined; + } finally { + clearTimeout(timer); + } + }, + now(): number { + return Date.now(); + }, + sleep(milliseconds: number): Promise { + return new Promise((resolve) => setTimeout(resolve, milliseconds)); + }, + // Not cryptographically secure, deliberately: the utilities that draw are generating + // example documents, the published package documents using \`Math.random()\` for exactly + // that, and \`crypto.getRandomValues\` measures 130x the cost per draw. A caller who needs + // unpredictability passes its own capability. + nextU32(): number { + return Math.floor(Math.random() * 4294967296); + }, + }; +} + +/** + * The platform default, built once at module load rather than per call — every public wrapper + * (\`docs/decisions/0011-public-entry-points-vs-capabilities.md\`) shares this one instance, the + * same way a caller who builds their own environment would share it across calls. + */ +export const DEFAULT_CAPABILITIES: Capabilities = defaultCapabilities(); +${race}`; + return { path: `capabilities${TYPESCRIPT_CONFIG.fileExtension}`, text }; +} + +function errorsModule(program: CProgram): { path: string; text: string } | undefined { + const declared = [...program.errors.values()]; + if (declared.length === 0) return undefined; + const lines = [ + "// Code generated by the logic engine. DO NOT EDIT.", + `// engine: ${ENGINE_VERSION}`, + "// source: errors", + "", + "/** The root of every domain error the core raises. */", + "export class DomainError extends Error {}", + "", + ]; + for (const error of declared) { + if (error.doc !== undefined) lines.push(`/** ${error.doc.split("\n")[0]} */`); + lines.push(`export class ${error.name} extends ${error.base ?? "DomainError"} {}`, ""); + } + return { path: `errors${TYPESCRIPT_CONFIG.fileExtension}`, text: lines.join("\n") }; +} + +import { ENGINE_VERSION } from "../../backend/generate.ts"; + + +/** + * The generated differential driver: one JSON line in, one JSON line out. The conformance harness + * speaks this protocol to every target, so the interpreter and the three backends are compared + * through exactly the same interface. + */ +function driverFiles(_program: CProgram, entries: readonly DriverEntry[]): { path: string; text: string }[] { + const imports = new Map(); + for (const entry of entries) { + const from = `./${entry.modulePath}`; + const names = imports.get(from) ?? []; + names.push(entry.targetName); + imports.set(from, names); + } + const lines = [ + "// Code generated by the logic engine. DO NOT EDIT.", + "// source: _driver", + "", + 'import { createInterface } from "node:readline";', + ]; + for (const [from, names] of imports) { + lines.push(`import { ${[...new Set(names)].sort().join(", ")} } from ${JSON.stringify(from)};`); + } + const needsEnv = entries.some((entry) => entry.usesEnv); + if (needsEnv) { + lines.push( + 'import { readFileSync, existsSync } from "node:fs";', + 'import { defaultCapabilities, type Capabilities, type HttpRequest, type HttpResponse } from "./capabilities.ts";', + "", + "/**", + " * The reference PCG32: same constants and default seed as `Interpreter`'s, so a draw", + " * matches the reference bit for bit. A fresh instance is built for every request, the same", + " * way the reference model starts a fresh interpreter — and so a fresh generator — per case.", + " */", + "class Pcg32 {", + "\tprivate state = 0n;", + "\tprivate readonly increment = 1442695040888963407n;", + "", + "\tconstructor(seed: bigint) {", + "\t\tthis.next();", + "\t\tthis.state = (this.state + seed) & 0xffffffffffffffffn;", + "\t\tthis.next();", + "\t}", + "", + "\tnext(): number {", + "\t\tconst previous = this.state;", + "\t\tthis.state = (previous * 6364136223846793005n + this.increment) & 0xffffffffffffffffn;", + "\t\tconst xorshifted = (((previous >> 18n) ^ previous) >> 27n) & 0xffffffffn;", + "\t\tconst rotation = previous >> 59n;", + "\t\treturn Number(((xorshifted >> rotation) | (xorshifted << ((-rotation) & 31n))) & 0xffffffffn);", + "\t}", + "}", + "", + "// The interpreter's own default: its constructor falls back to this seed whenever", + "// `Capabilities.seed` is left unset, which is how every conformance case runs it.", + "const DEFAULT_SEED = 0x853c49e6748fea9bn;", + "", + "/**", + " * The capability fake the differential harness drives.", + " *", + " * Responses come from `fixtures.json`, a URL that is not in it models a transport error,", + " * and the scripted latency is what decides a race, the same way the reference model's", + " * virtual clock decides it. `nextU32` gets a fresh PCG32 per call, matching the reference", + " * model's fresh interpreter per case.", + " */", + "type Fixture = { status: number; body: string; latencyMillis?: number };", + "", + "function fakeCapabilities(fixtures: Record): Capabilities {", + "\tconst random = new Pcg32(DEFAULT_SEED);", + "", + "\treturn {", + "\t\tasync request(request: HttpRequest): Promise {", + "\t\t\tconst fixture = fixtures[request.url];", + "", + "\t\t\tif (fixture === undefined) return undefined;", + "", + "\t\t\tawait new Promise((resolve) => setTimeout(resolve, fixture.latencyMillis ?? 0));", + "", + "\t\t\treturn { status: fixture.status, headers: [], body: fixture.body };", + "\t\t},", + "\t\tnow: () => 0,", + "\t\tsleep: (milliseconds: number) => new Promise((resolve) => setTimeout(resolve, milliseconds)),", + "\t\tnextU32: () => random.next(),", + "\t};", + "}", + ); + } + lines.push( + "", + "type Handler = (args: readonly unknown[], env: unknown) => unknown;", + "", + "const handlers: Record = {", + ); + for (const entry of entries) { + // `Parameters` keeps the driver free of type imports and still type checks. + const args = entry.params.map( + (_param, index) => `args[${index}] as Parameters[${index}]`, + ); + if (entry.usesEnv) args.push("env as Capabilities"); + lines.push(`\t${JSON.stringify(entry.coreName)}: (args, env) => ${entry.targetName}(${args.join(", ")}),`); + } + lines.push( + "};", + "", + ...(needsEnv + ? [ + 'const fixturePath = new URL("fixtures.json", import.meta.url).pathname;', + "const fixtures = existsSync(fixturePath)\n\t\t\t? (JSON.parse(readFileSync(fixturePath, \"utf8\")) as Record)\n\t\t\t: undefined;", + "", + ] + : []), + "const reader = createInterface({ input: process.stdin });", + "", + "for await (const line of reader) {", + '\tif (line.trim() === "") continue;', + "\tconst request = JSON.parse(line) as { fn: string; args: unknown[] };", + "", + ...(needsEnv + ? [ + "\t// A fresh environment per line: nextU32 starts from the same state the reference model's", + "\t// fresh interpreter starts from for every case.", + "\tconst environment = fixtures === undefined ? defaultCapabilities() : fakeCapabilities(fixtures);", + "", + ] + : []), + "\ttry {", + `\t\tconst value = await handlers[request.fn]!(request.args, ${needsEnv ? "environment" : "undefined"});`, + '\t\tprocess.stdout.write(`${JSON.stringify({ ok: true, value: value === undefined ? null : value })}\\n`);', + "\t} catch (error) {", + '\t\tprocess.stdout.write(`${JSON.stringify({ ok: false, error: (error as Error).constructor.name })}\\n`);', + "\t}", + "}", + "", + ); + return [{ path: "_driver.ts", text: lines.join("\n") }]; +} + +export const TYPESCRIPT_BACKEND: Backend = { + spec: TYPESCRIPT_SPEC, + fileExtension: TYPESCRIPT_CONFIG.fileExtension, + printModule, + importPath, + support: supportModule, + errorsModule, + renderType: tsType, + comment: "//", + driver: driverFiles, + supportImport: (needs, usesEnv, builtins) => { + const types = [...builtins]; + if (usesEnv && needs.env) types.push("Capabilities"); + const values = usesEnv && needs.race ? ["raceFirstSome"] : []; + return [ + { from: "SUPPORT", names: types.sort(), typeOnly: true }, + { from: "SUPPORT", names: values }, + ]; + }, + defaultCapabilities: { + ref: { kind: "name", name: "DEFAULT_CAPABILITIES" }, + imports: [{ from: "SUPPORT", names: ["DEFAULT_CAPABILITIES"] }], + seamName: (publicName) => `${publicName}With`, + }, + // Measured, not assumed: V8 already inlines a small monomorphic call once it is hot, so this + // engine's own call-site inlining (`optimize/inline.ts`) is set conservatively here — see + // `engine/docs/progress.md` §8 for the before/after and the bundle-size effect this was weighed + // against. + inlineBudget: { maxStatements: 8, rounds: 4, maxDuplicatedNodes: 6 }, +}; diff --git a/engine/src/types.ts b/engine/src/types.ts new file mode 100644 index 000000000..eee03e253 --- /dev/null +++ b/engine/src/types.ts @@ -0,0 +1,340 @@ +/** + * Semantic types: what a value means, independent of any target language. + * + * Types carry their refinements (an integer's proven range, a string's character class and + * length range, a list's length range). Refinements are what make an idiomatic native lowering + * provably safe, so they live in the type rather than in a side table. + */ + +/** + * The largest number of elements a list or scalars a string may hold. Every collection length and + * every index derived from one is bounded by this constant, which is how the analysis keeps every + * integer range finite without asking authors to annotate lengths. It is the documented platform + * limit shared by the three targets (see docs/semantics.md, "Integers"). + */ +export const MAX_COLLECTION_LENGTH = 2 ** 31 - 1; + +/** The integer domain every target represents exactly with its default integer type. */ +export const SAFE_INT_LO = -(2n ** 53n - 1n); +export const SAFE_INT_HI = 2n ** 53n - 1n; + +/** Character classes a `String` can be refined to. `digits` implies `ascii`. */ +export type StringClass = "none" | "ascii" | "digits"; + +export type SemType = + | { readonly kind: "Bool" } + | { readonly kind: "Int"; readonly lo: bigint; readonly hi: bigint } + | { readonly kind: "Float" } + | { readonly kind: "Decimal"; readonly scale: number } + | { + readonly kind: "String"; + readonly cls: StringClass; + readonly min: number; + readonly max: number; + /** Source text of a regex the value is proven to match, when known. */ + readonly pattern?: string; + } + | { readonly kind: "List"; readonly elem: SemType; readonly min: number; readonly max: number } + | { readonly kind: "Option"; readonly inner: SemType } + | { readonly kind: "Record"; readonly name: string } + | { readonly kind: "Enum"; readonly name: string; readonly members: readonly string[] } + | { readonly kind: "Union"; readonly name: string } + | { readonly kind: "CivilDate" } + | { readonly kind: "Instant" } + | { readonly kind: "Duration" } + | { readonly kind: "Lambda"; readonly params: readonly SemType[]; readonly ret: SemType } + | { readonly kind: "Void" } + /** The type of an expression that never produces a value (a `throw`, or a diverging branch). */ + | { readonly kind: "Never" }; + +export const tBool: SemType = { kind: "Bool" }; +export const tFloat: SemType = { kind: "Float" }; +export const tVoid: SemType = { kind: "Void" }; +export const tNever: SemType = { kind: "Never" }; +export const tCivilDate: SemType = { kind: "CivilDate" }; +export const tInstant: SemType = { kind: "Instant" }; +export const tDuration: SemType = { kind: "Duration" }; + +export function tInt(lo: bigint, hi: bigint): SemType { + return { kind: "Int", lo, hi }; +} + +/** The default integer domain of an unannotated `Int`. */ +export function tIntDefault(): SemType { + return tInt(SAFE_INT_LO, SAFE_INT_HI); +} + +/** The range of any collection length or index. */ +export function tIndex(): SemType { + return tInt(0n, BigInt(MAX_COLLECTION_LENGTH)); +} + +export function tIntLit(value: bigint): SemType { + return tInt(value, value); +} + +export function tDecimal(scale: number): SemType { + return { kind: "Decimal", scale }; +} + +export function tString( + cls: StringClass = "none", + min = 0, + max = MAX_COLLECTION_LENGTH, + pattern?: string, +): SemType { + return { kind: "String", cls, min, max, pattern }; +} + +export function tList(elem: SemType, min = 0, max = MAX_COLLECTION_LENGTH): SemType { + return { kind: "List", elem, min, max }; +} + +export function tOption(inner: SemType): SemType { + return { kind: "Option", inner }; +} + +export function tRecord(name: string): SemType { + return { kind: "Record", name }; +} + +export function tEnum(name: string, members: readonly string[]): SemType { + return { kind: "Enum", name, members }; +} + +export function tUnion(name: string): SemType { + return { kind: "Union", name }; +} + +export function tLambda(params: readonly SemType[], ret: SemType): SemType { + return { kind: "Lambda", params, ret }; +} + +const CLASS_ORDER: Record = { none: 0, ascii: 1, digits: 2 }; + +/** True when `cls` guarantees at least what `required` guarantees. */ +export function classSatisfies(cls: StringClass, required: StringClass): boolean { + return CLASS_ORDER[cls] >= CLASS_ORDER[required]; +} + +export function weakerClass(a: StringClass, b: StringClass): StringClass { + return CLASS_ORDER[a] <= CLASS_ORDER[b] ? a : b; +} + +export function strongerClass(a: StringClass, b: StringClass): StringClass { + return CLASS_ORDER[a] >= CLASS_ORDER[b] ? a : b; +} + +/** `a <: b`: every value of `a` is a value of `b`. Refinements narrow, so they are subtypes. */ +export function isSubtype(a: SemType, b: SemType): boolean { + if (a.kind === "Never") return true; + if (a.kind !== b.kind) { + // A refined value flows into an Option without ceremony only through `some`, never here. + return false; + } + + switch (a.kind) { + case "Bool": + case "Float": + case "CivilDate": + case "Instant": + case "Duration": + case "Void": + return true; + case "Int": { + const other = b as Extract; + return other.lo <= a.lo && a.hi <= other.hi; + } + case "Decimal": + return a.scale === (b as Extract).scale; + case "String": { + const other = b as Extract; + if (!classSatisfies(a.cls, other.cls)) return false; + if (other.pattern !== undefined && other.pattern !== a.pattern) return false; + return other.min <= a.min && a.max <= other.max; + } + case "List": { + const other = b as Extract; + return ( + isSubtype(a.elem, other.elem) && other.min <= a.min && a.max <= other.max + ); + } + case "Option": + return isSubtype(a.inner, (b as Extract).inner); + case "Record": + return a.name === (b as Extract).name; + case "Union": + return a.name === (b as Extract).name; + case "Enum": { + const other = b as Extract; + return a.members.every((member) => other.members.includes(member)); + } + case "Lambda": { + const other = b as Extract; + return ( + a.params.length === other.params.length && + a.params.every((param, index) => isSubtype(other.params[index]!, param)) && + isSubtype(a.ret, other.ret) + ); + } + default: { + const exhaustive: never = a; + return exhaustive; + } + } +} + +/** Least upper bound of two types, used when control flow merges. */ +export function join(a: SemType, b: SemType): SemType { + if (a.kind === "Never") return b; + if (b.kind === "Never") return a; + if (a.kind === "Option" && b.kind !== "Option") return tOption(join(a.inner, b)); + if (b.kind === "Option" && a.kind !== "Option") return tOption(join(a, b.inner)); + if (a.kind !== b.kind) return a; + + switch (a.kind) { + case "Int": { + const other = b as Extract; + return tInt(a.lo < other.lo ? a.lo : other.lo, a.hi > other.hi ? a.hi : other.hi); + } + case "String": { + const other = b as Extract; + return tString( + weakerClass(a.cls, other.cls), + Math.min(a.min, other.min), + Math.max(a.max, other.max), + a.pattern === other.pattern ? a.pattern : undefined, + ); + } + case "List": { + const other = b as Extract; + return tList(join(a.elem, other.elem), Math.min(a.min, other.min), Math.max(a.max, other.max)); + } + case "Option": + return tOption(join(a.inner, (b as Extract).inner)); + case "Enum": { + const other = b as Extract; + const members = [...new Set([...a.members, ...other.members])].sort(); + return tEnum(a.name, members); + } + default: + return a; + } +} + +/** Drops refinements, keeping only the shape. Used to compare declared and inferred types. */ +export function widen(type: SemType): SemType { + switch (type.kind) { + case "Int": + return tIntDefault(); + case "String": + return tString("none"); + case "List": + return tList(widen(type.elem)); + case "Option": + return tOption(widen(type.inner)); + default: + return type; + } +} + +export function typeToString(type: SemType): string { + switch (type.kind) { + case "Bool": + return "Bool"; + case "Int": + return `Int[${type.lo}..${type.hi}]`; + case "Float": + return "Float"; + case "Decimal": + return `Decimal<${type.scale}>`; + case "String": { + const base = type.cls === "none" ? "String" : type.cls === "ascii" ? "Ascii" : "Digits"; + const length = type.min === type.max ? `${type.min}` : `${type.min}..${type.max}`; + const pattern = type.pattern === undefined ? "" : ` matches ${type.pattern}`; + return `${base}[${length}]${pattern}`; + } + case "List": + return `List<${typeToString(type.elem)}>[${type.min}..${type.max}]`; + case "Option": + return `Option<${typeToString(type.inner)}>`; + case "Record": + return type.name; + case "Union": + return type.name; + case "Enum": + return type.members.map((member) => JSON.stringify(member)).join(" | "); + case "CivilDate": + return "CivilDate"; + case "Instant": + return "Instant"; + case "Duration": + return "Duration"; + case "Lambda": + return `(${type.params.map(typeToString).join(", ")}) => ${typeToString(type.ret)}`; + case "Void": + return "Void"; + case "Never": + return "Never"; + default: { + const exhaustive: never = type; + return exhaustive; + } + } +} + +/* ------------------------------------------------------------------ * + * Interval algebra over proven integer ranges. + * ------------------------------------------------------------------ */ + +export type Range = { readonly lo: bigint; readonly hi: bigint }; + +export function rangeOf(type: SemType): Range { + return type.kind === "Int" ? { lo: type.lo, hi: type.hi } : { lo: SAFE_INT_LO, hi: SAFE_INT_HI }; +} + +const min = (...values: bigint[]): bigint => values.reduce((a, b) => (a < b ? a : b)); +const max = (...values: bigint[]): bigint => values.reduce((a, b) => (a > b ? a : b)); + +export function rangeAdd(a: Range, b: Range): Range { + return { lo: a.lo + b.lo, hi: a.hi + b.hi }; +} + +export function rangeSub(a: Range, b: Range): Range { + return { lo: a.lo - b.hi, hi: a.hi - b.lo }; +} + +export function rangeMul(a: Range, b: Range): Range { + const products = [a.lo * b.lo, a.lo * b.hi, a.hi * b.lo, a.hi * b.hi]; + return { lo: min(...products), hi: max(...products) }; +} + +/** Truncated division, the semantics of `/` and `%` in the Core (see docs/semantics.md). */ +export function rangeDiv(a: Range, b: Range): Range { + const divisors = [b.lo, b.hi].filter((value) => value !== 0n); + if (divisors.length === 0) return { lo: 0n, hi: 0n }; + const quotients: bigint[] = []; + for (const divisor of divisors) { + quotients.push(a.lo / divisor, a.hi / divisor); + } + return { lo: min(...quotients), hi: max(...quotients) }; +} + +export function rangeMod(a: Range, b: Range): Range { + const bound = max(b.lo < 0n ? -b.lo : b.lo, b.hi < 0n ? -b.hi : b.hi) - 1n; + const lo = a.lo < 0n ? -bound : 0n; + const hi = a.hi > 0n ? bound : 0n; + return { lo, hi: hi < lo ? lo : hi }; +} + +export function rangeUnion(a: Range, b: Range): Range { + return { lo: min(a.lo, b.lo), hi: max(a.hi, b.hi) }; +} + +export function rangeContains(outer: Range, inner: Range): boolean { + return outer.lo <= inner.lo && inner.hi <= outer.hi; +} + +export function rangeIsConstant(range: Range): boolean { + return range.lo === range.hi; +} diff --git a/engine/src/values.ts b/engine/src/values.ts new file mode 100644 index 000000000..da3f1b81c --- /dev/null +++ b/engine/src/values.ts @@ -0,0 +1,237 @@ +/** + * Runtime values of the reference semantics. + * + * The interpreter, the comptime evaluator and the conformance harness all speak this + * representation. It mirrors the semantic types one to one, so a divergence between the + * interpreter and a generated target is always a compiler bug, never a representation accident. + */ + +export type RecordValue = { + readonly __kind: "record"; + readonly type: string; + readonly fields: Readonly>; +}; + +export type DecimalValue = { + readonly __kind: "decimal"; + readonly unscaled: bigint; + readonly scale: number; +}; + +export type DateValue = { readonly __kind: "date"; readonly days: number }; + +export type NoneValue = { readonly __kind: "none" }; + +export type SomeValue = { readonly __kind: "some"; readonly value: Value }; + +export type LambdaValue = { + readonly __kind: "lambda"; + readonly call: (args: Value[]) => Value; +}; + +export type Value = + | boolean + | bigint + | number + | string + | readonly Value[] + | RecordValue + | DecimalValue + | DateValue + | NoneValue + | SomeValue + | LambdaValue; + +export const NONE: NoneValue = { __kind: "none" }; + +export function some(value: Value): SomeValue { + return { __kind: "some", value }; +} + +export function isNone(value: Value): boolean { + return typeof value === "object" && value !== null && "__kind" in value && value.__kind === "none"; +} + +export function unwrap(value: Value): Value { + if (typeof value === "object" && value !== null && "__kind" in value && value.__kind === "some") { + return value.value; + } + throw new Error("unwrap of a none value"); +} + +export function record(type: string, fields: Record): RecordValue { + return { __kind: "record", type, fields }; +} + +export function decimal(unscaled: bigint, scale: number): DecimalValue { + return { __kind: "decimal", unscaled, scale }; +} + +export function civilDate(days: number): DateValue { + return { __kind: "date", days }; +} + +export function asBigInt(value: Value): bigint { + if (typeof value === "bigint") return value; + throw new Error(`expected an integer, got ${typeof value}`); +} + +export function asNumber(value: Value): number { + if (typeof value === "number") return value; + throw new Error(`expected a float, got ${typeof value}`); +} + +export function asString(value: Value): string { + if (typeof value === "string") return value; + throw new Error(`expected a string, got ${typeof value}`); +} + +export function asBool(value: Value): boolean { + if (typeof value === "boolean") return value; + throw new Error(`expected a boolean, got ${typeof value}`); +} + +export function asList(value: Value): readonly Value[] { + if (Array.isArray(value)) return value as readonly Value[]; + throw new Error("expected a list"); +} + +export function asRecord(value: Value): RecordValue { + if (typeof value === "object" && value !== null && "__kind" in value && value.__kind === "record") { + return value; + } + throw new Error("expected a record"); +} + +export function asDecimal(value: Value): DecimalValue { + if (typeof value === "object" && value !== null && "__kind" in value && value.__kind === "decimal") { + return value; + } + throw new Error("expected a decimal"); +} + +export function asDate(value: Value): DateValue { + if (typeof value === "object" && value !== null && "__kind" in value && value.__kind === "date") { + return value; + } + throw new Error("expected a date"); +} + +export function asLambda(value: Value): LambdaValue { + if (typeof value === "object" && value !== null && "__kind" in value && value.__kind === "lambda") { + return value; + } + throw new Error("expected a lambda"); +} + +/** The scalars of a string, as code points. `String#length` in the semantics counts these. */ +export function codePointsOf(text: string): number[] { + return [...text].map((scalar) => scalar.codePointAt(0)!); +} + +export function fromCodePoints(points: readonly number[]): string { + return points.map((point) => String.fromCodePoint(point)).join(""); +} + +/** Scalar-order comparison, the semantics of `str.compare` in every target. */ +export function compareScalars(a: string, b: string): number { + const left = codePointsOf(a); + const right = codePointsOf(b); + const shared = Math.min(left.length, right.length); + for (let index = 0; index < shared; index++) { + if (left[index]! !== right[index]!) return left[index]! < right[index]! ? -1 : 1; + } + return left.length === right.length ? 0 : left.length < right.length ? -1 : 1; +} + +/** Structural equality over values, used by `===` in the Core and by the conformance differ. */ +export function valuesEqual(a: Value, b: Value): boolean { + if (typeof a !== typeof b) return false; + if (Array.isArray(a) && Array.isArray(b)) { + return a.length === b.length && a.every((item, index) => valuesEqual(item, b[index]!)); + } + if (typeof a === "object" && typeof b === "object" && a !== null && b !== null) { + const left = a as { __kind: string }; + const right = b as { __kind: string }; + if (left.__kind !== right.__kind) return false; + switch (left.__kind) { + case "none": + return true; + case "some": + return valuesEqual((left as SomeValue).value, (right as SomeValue).value); + case "decimal": { + const x = left as DecimalValue; + const y = right as DecimalValue; + return x.scale === y.scale && x.unscaled === y.unscaled; + } + case "date": + return (left as DateValue).days === (right as DateValue).days; + case "record": { + const x = left as RecordValue; + const y = right as RecordValue; + const keys = Object.keys(x.fields); + return ( + x.type === y.type && + keys.length === Object.keys(y.fields).length && + keys.every((key) => valuesEqual(x.fields[key]!, y.fields[key]!)) + ); + } + default: + return false; + } + } + return a === b; +} + +/** A stable, language-neutral JSON encoding used by the conformance protocol. */ +export function encodeValue(value: Value): unknown { + if (typeof value === "bigint") return { $int: value.toString() }; + if (Array.isArray(value)) return (value as readonly Value[]).map(encodeValue); + if (typeof value === "object" && value !== null) { + const tagged = value as { __kind: string }; + switch (tagged.__kind) { + case "none": + return { $none: true }; + case "some": + return { $some: encodeValue((tagged as SomeValue).value) }; + case "decimal": { + const item = tagged as DecimalValue; + return { $dec: item.unscaled.toString(), scale: item.scale }; + } + case "date": + return { $date: (tagged as DateValue).days }; + case "record": { + const item = tagged as RecordValue; + const fields: Record = {}; + for (const key of Object.keys(item.fields).sort()) { + fields[key] = encodeValue(item.fields[key]!); + } + return { $record: item.type, fields }; + } + default: + return null; + } + } + return value; +} + +export function decodeValue(raw: unknown): Value { + if (raw === null) return NONE; + if (Array.isArray(raw)) return raw.map(decodeValue); + if (typeof raw === "object") { + const item = raw as Record; + if ("$int" in item) return BigInt(item["$int"] as string); + if ("$none" in item) return NONE; + if ("$some" in item) return some(decodeValue(item["$some"])); + if ("$dec" in item) return decimal(BigInt(item["$dec"] as string), item["scale"] as number); + if ("$date" in item) return civilDate(item["$date"] as number); + if ("$record" in item) { + const fields: Record = {}; + for (const [key, value] of Object.entries(item["fields"] as Record)) { + fields[key] = decodeValue(value); + } + return record(item["$record"] as string, fields); + } + } + return raw as Value; +} diff --git a/engine/stdlib/date.ts b/engine/stdlib/date.ts new file mode 100644 index 000000000..076f3ea01 --- /dev/null +++ b/engine/stdlib/date.ts @@ -0,0 +1,101 @@ +/** + * Civil date arithmetic, written once in the source language. + * + * `CivilDate` is days since 1970-01-01 in every target, so the calendar itself never depends on a + * host library: the algorithms below (Howard Hinnant's `days_from_civil` and `civil_from_days`) + * are exact for the proleptic Gregorian calendar and identical everywhere. + */ + +/** Floor division, which the calendar algorithms need for negative years. */ +export function floorDiv(value: IntRange<-4000000, 4000000>, divisor: IntRange<1, 146097>): Int { + const quotient = value / divisor; + + if (value < 0 && quotient * divisor !== value) { + return quotient - 1; + } + + return quotient; +} + +/** Days since 1970-01-01 for a year, month and day already known to be a real date. */ +export function daysFromCivil( + year: IntRange<1, 9999>, + month: IntRange<1, 12>, + day: IntRange<1, 31>, +): IntRange<-719162, 2932896> { + const shifted = month <= 2 ? year - 1 : year; + const era = floorDiv(shifted, 400); + const yearOfEra = shifted - era * 400; + const monthTerm = month > 2 ? month - 3 : month + 9; + const dayOfYear = (153 * monthTerm + 2) / 5 + day - 1; + const dayOfEra = yearOfEra * 365 + yearOfEra / 4 - yearOfEra / 100 + dayOfYear; + + // Interval analysis cannot prove the calendar identity behind this algorithm, so the result is + // clamped to the range the type promises. For a real date the clamp never fires; for anything + // else it keeps the function total, which is what lets every caller stay provably safe. + return int.min(int.max(era * 146097 + dayOfEra - 719468, -719162), 2932896); +} + +/** The year of a date given as days since 1970-01-01. */ +export function yearFromDays(days: IntRange<-719162, 2932896>): IntRange<1, 9999> { + const shifted = days + 719468; + const era = floorDiv(shifted, 146097); + const dayOfEra = shifted - era * 146097; + const yearOfEra = (dayOfEra - dayOfEra / 1460 + dayOfEra / 36524 - dayOfEra / 146096) / 365; + const year = yearOfEra + era * 400; + const dayOfYear = dayOfEra - (365 * yearOfEra + yearOfEra / 4 - yearOfEra / 100); + const monthPrime = (5 * dayOfYear + 2) / 153; + const month = monthPrime < 10 ? monthPrime + 3 : monthPrime - 9; + + return int.min(int.max(month <= 2 ? year + 1 : year, 1), 9999); +} + +/** The month of a date given as days since 1970-01-01. */ +export function monthFromDays(days: IntRange<-719162, 2932896>): IntRange<1, 12> { + const shifted = days + 719468; + const era = floorDiv(shifted, 146097); + const dayOfEra = shifted - era * 146097; + const yearOfEra = (dayOfEra - dayOfEra / 1460 + dayOfEra / 36524 - dayOfEra / 146096) / 365; + const dayOfYear = dayOfEra - (365 * yearOfEra + yearOfEra / 4 - yearOfEra / 100); + const monthPrime = (5 * dayOfYear + 2) / 153; + + return int.min(int.max(monthPrime < 10 ? monthPrime + 3 : monthPrime - 9, 1), 12); +} + +/** The day of month of a date given as days since 1970-01-01. */ +export function dayFromDays(days: IntRange<-719162, 2932896>): IntRange<1, 31> { + const shifted = days + 719468; + const era = floorDiv(shifted, 146097); + const dayOfEra = shifted - era * 146097; + const yearOfEra = (dayOfEra - dayOfEra / 1460 + dayOfEra / 36524 - dayOfEra / 146096) / 365; + const dayOfYear = dayOfEra - (365 * yearOfEra + yearOfEra / 4 - yearOfEra / 100); + const monthPrime = (5 * dayOfYear + 2) / 153; + + return int.min(int.max(dayOfYear - (153 * monthPrime + 2) / 5 + 1, 1), 31); +} + +/** Whether a year, month and day name a real date on the proleptic Gregorian calendar. */ +export function isRealDate(year: Int, month: Int, day: Int): boolean { + return ymdToDays(year, month, day) !== undefined; +} + +/** + * Days since 1970-01-01, or absent when the components do not name a real date. + * + * The bounds are checked here rather than in a helper because the checker reads a guard, not a + * called predicate: after this `if`, the three components carry the ranges `daysFromCivil` + * requires, and the round trip rejects a day the month does not have. + */ +export function ymdToDays(year: Int, month: Int, day: Int): Int | undefined { + if (year < 1 || year > 9999 || month < 1 || month > 12 || day < 1 || day > 31) { + return undefined; + } + + const days = daysFromCivil(year, month, day); + + if (yearFromDays(days) !== year || monthFromDays(days) !== month || dayFromDays(days) !== day) { + return undefined; + } + + return days; +} diff --git a/engine/stdlib/strings.ts b/engine/stdlib/strings.ts new file mode 100644 index 000000000..61bce533a --- /dev/null +++ b/engine/stdlib/strings.ts @@ -0,0 +1,70 @@ +/** + * Portable string operations. + * + * These exist because at least one target cannot lower the operation natively with proven + * equivalence. `compareScalars` is the clearest case: JavaScript compares strings by UTF-16 code + * unit, so a non-ASCII value would sort differently there than in Python or Go, and the portable + * implementation is selected unless the value is proven ASCII. + */ + +/** Scalar-order comparison: -1, 0 or 1, identical in every target. */ +export function compareScalars(left: string, right: string): IntRange<-1, 1> { + const leftPoints = str.codePoints(left); + const rightPoints = str.codePoints(right); + const shared = int.min(leftPoints.length, rightPoints.length); + + for (let index = 0; index < shared; index++) { + // The bound comes from a length neither list's type records, so the checked accessor is + // what keeps the access provably safe. + const a = seq.at(leftPoints, index) ?? 0; + const b = seq.at(rightPoints, index) ?? 0; + + if (a < b) { + return -1; + } + + if (a > b) { + return 1; + } + } + + if (leftPoints.length < rightPoints.length) { + return -1; + } + + if (leftPoints.length > rightPoints.length) { + return 1; + } + + return 0; +} + +/** ASCII-only upper casing, for values that are not proven ASCII. */ +export function asciiUpperAll(value: string): string { + let points: IntRange<0, 1114111>[] = []; + + for (const point of str.codePoints(value)) { + if (point >= 97 && point <= 122) { + points.push(point - 32); + } else { + points.push(point); + } + } + + return str.fromCodePoints(points); +} + +/** ASCII-only lower casing, for values that are not proven ASCII. */ +export function asciiLowerAll(value: string): string { + let points: IntRange<0, 1114111>[] = []; + + for (const point of str.codePoints(value)) { + if (point >= 65 && point <= 90) { + points.push(point + 32); + } else { + points.push(point); + } + } + + return str.fromCodePoints(points); +} diff --git a/engine/tests/boundary.spec.ts b/engine/tests/boundary.spec.ts new file mode 100644 index 000000000..1f939d2c3 --- /dev/null +++ b/engine/tests/boundary.spec.ts @@ -0,0 +1,88 @@ +/** + * The import boundaries the architecture depends on. + * + * Nothing after the HIR may know about the parser, and nothing before the backends may know about + * a target. Both are checked by reading the sources rather than by convention, because a violation + * is exactly the kind of thing that creeps in one import at a time. + */ + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync, readdirSync, statSync } from "node:fs"; +import { join, relative } from "node:path"; + +const SRC = join(import.meta.dirname, "..", "src"); + +function sourceFiles(root: string): string[] { + const files: string[] = []; + const walk = (directory: string): void => { + for (const entry of readdirSync(directory)) { + const full = join(directory, entry); + if (statSync(full).isDirectory()) walk(full); + else if (entry.endsWith(".ts")) files.push(full); + } + }; + walk(root); + return files; +} + +const FRONTEND_ONLY = ["src/frontend/lower.ts"]; + +test("only the frontend imports the parser", () => { + for (const file of sourceFiles(SRC)) { + const relativePath = relative(join(SRC, ".."), file).split("\\").join("/"); + if (FRONTEND_ONLY.includes(relativePath)) continue; + const source = readFileSync(file, "utf8"); + assert.ok( + !source.includes('from "oxc-parser"'), + `${relativePath} imports the parser; only the frontend may`, + ); + } +}); + +const TARGET_FREE_DIRECTORIES = [ + "frontend", + "hir", + "core", + "link", + "comptime", + "analysis", + "optimize", + "interp", + "intrinsics", +]; + +test("nothing before the backends imports a target", () => { + for (const directory of TARGET_FREE_DIRECTORIES) { + for (const file of sourceFiles(join(SRC, directory))) { + const source = readFileSync(file, "utf8"); + assert.ok( + !source.includes("targets/"), + `${relative(SRC, file)} imports a target; lowering decisions belong in a backend`, + ); + } + } +}); + +test("nothing before the backends branches on a target name", () => { + const names = ["typescript", "python", "golang"]; + for (const directory of TARGET_FREE_DIRECTORIES) { + for (const file of sourceFiles(join(SRC, directory))) { + const source = readFileSync(file, "utf8").toLowerCase(); + for (const name of names) { + assert.ok( + !source.includes(`=== "${name}"`), + `${relative(SRC, file)} branches on the ${name} target`, + ); + } + } + } +}); + +test("the standard library only uses the source subset", () => { + const stdlib = join(import.meta.dirname, "..", "stdlib"); + for (const file of sourceFiles(stdlib)) { + const source = readFileSync(file, "utf8"); + assert.ok(!source.includes("import "), `${relative(stdlib, file)} imports; the standard library is self-contained`); + } +}); diff --git a/engine/tests/determinism.spec.ts b/engine/tests/determinism.spec.ts new file mode 100644 index 000000000..9be5d02c8 --- /dev/null +++ b/engine/tests/determinism.spec.ts @@ -0,0 +1,39 @@ +/** + * Output has to be byte-identical across runs: a golden file that changes because a Map iterated + * differently is a golden file nobody trusts. + */ + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { join } from "node:path"; +import { compileProject } from "../src/api.ts"; +import { generate } from "../src/backend/generate.ts"; +import { GO_BACKEND } from "../src/targets/go/index.ts"; +import { PYTHON_BACKEND } from "../src/targets/python/index.ts"; +import { TYPESCRIPT_BACKEND } from "../src/targets/typescript/index.ts"; +import { RUST_BACKEND } from "../src/targets/rust/index.ts"; + +const EXAMPLE = join(import.meta.dirname, "..", "examples", "generic", "source"); + +test("generating twice produces identical bytes", () => { + for (const backend of [TYPESCRIPT_BACKEND, PYTHON_BACKEND, GO_BACKEND, RUST_BACKEND]) { + const first = generate(compileProject(EXAMPLE).program, backend); + const second = generate(compileProject(EXAMPLE).program, backend); + assert.deepEqual( + first.files.map((file) => [file.path, file.text]), + second.files.map((file) => [file.path, file.text]), + `${backend.spec.name} is not deterministic`, + ); + assert.equal(first.lowering, second.lowering, `${backend.spec.name} LOWERING.md is not deterministic`); + } +}); + +test("the idiomatic and the plain form generate the same module set", () => { + const compilation = compileProject(EXAMPLE); + const idiomatic = generate(compilation.program, TYPESCRIPT_BACKEND); + const plain = generate(compilation.program, TYPESCRIPT_BACKEND, { noIdioms: true }); + assert.deepEqual( + idiomatic.files.map((file) => file.path).sort(), + plain.files.map((file) => file.path).sort(), + ); +}); diff --git a/engine/tests/example.spec.ts b/engine/tests/example.spec.ts new file mode 100644 index 000000000..c41107ae7 --- /dev/null +++ b/engine/tests/example.spec.ts @@ -0,0 +1,59 @@ +/** + * The engine is generic. + * + * `examples/generic` is a project with no relation to the one that motivated the engine: it is + * compiled by the same compiler, with the same standard library, and generates the same four + * targets. This test is what keeps a Brazilian Utils assumption from leaking into the compiler. + */ + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { join } from "node:path"; +import { compileProject } from "../src/api.ts"; +import { generate } from "../src/backend/generate.ts"; +import { Interpreter } from "../src/interp/interp.ts"; +import { TYPESCRIPT_BACKEND } from "../src/targets/typescript/index.ts"; +import { PYTHON_BACKEND } from "../src/targets/python/index.ts"; +import { GO_BACKEND } from "../src/targets/go/index.ts"; +import { RUST_BACKEND } from "../src/targets/rust/index.ts"; + +const EXAMPLE = join(import.meta.dirname, "..", "examples", "generic", "source"); + +test("a project with no Brazilian anything compiles and runs", () => { + const compilation = compileProject(EXAMPLE); + const interpreter = new Interpreter(compilation.program); + + for (const [value, expected] of [ + ["4539578763621486", true], + ["4539578763621487", false], + ["79927398713", true], + ["79927398710", false], + ["", false], + ["1", false], + ["abc", false], + ] as const) { + assert.equal(interpreter.call("is-valid-luhn::isValidLuhn", [value]), expected, value); + } + + for (const [value, expected] of [ + ["Hello, World!", "hello-world"], + [" spaced out ", "spaced-out"], + ["ALREADY-slug-99", "already-slug-99"], + ["", ""], + ] as const) { + assert.equal(interpreter.call("slugify::slugify", [value]), expected, value); + } +}); + +test("the example generates for every target", () => { + const compilation = compileProject(EXAMPLE); + for (const backend of [TYPESCRIPT_BACKEND, PYTHON_BACKEND, GO_BACKEND, RUST_BACKEND]) { + const result = generate(compilation.program, backend); + assert.ok(result.files.length > 0, `${backend.spec.name} generated nothing`); + for (const file of result.files) { + // A Python package marker is deliberately empty; everything else has content. + if (file.path.endsWith("__init__.py")) continue; + assert.ok(file.text.length > 0, `${file.path} is empty`); + } + } +}); diff --git a/engine/tests/fixtures/example.LOWERING.md b/engine/tests/fixtures/example.LOWERING.md new file mode 100644 index 000000000..777d33088 --- /dev/null +++ b/engine/tests/fixtures/example.LOWERING.md @@ -0,0 +1,34 @@ +# Lowering selections — typescript + +Generated by the engine. Each row is one operation, the argument types it was called +with, the implementation that was selected, and the rule that decided it. + +| operation | argument types | implementation | why | +| --- | --- | --- | --- | +| `core.eq` | `Int[-1..1], Int[0..0]` | native | only candidate, cost none/constant | +| `core.eq` | `Int[0..9], Int[0..0]` | native | only candidate, cost none/constant | +| `int.add` | `Int[0..162], Int[0..9]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..1114111], Int[48..48]` | native | only candidate, cost none/constant | +| `int.ge` | `Int[0..1114111], Int[97..97]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[0..18], Int[9..9]` | native | only candidate, cost none/constant | +| `int.gt` | `Int[0..2147483647], Int[0..0]` | native | only candidate, cost none/constant | +| `int.le` | `Int[48..1114111], Int[57..57]` | native | only candidate, cost none/constant | +| `int.le` | `Int[97..1114111], Int[122..122]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[-16..19], Int[2..2]` | native | only candidate, cost none/constant | +| `int.mod` | `Int[0..171], Int[10..10]` | native | only candidate, cost none/constant | +| `int.mul` | `Int[0..9], Int[2..2]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[10..18], Int[9..9]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[2..19], Int[0..18]` | native | only candidate, cost none/constant | +| `int.sub` | `Int[48..57], Int[48..48]` | native | only candidate, cost none/constant | +| `opt.orElse` | `Option, Int[48..48]` | native | only candidate, cost none/constant | +| `re.test` | `String[0..2147483647]` | native | only candidate, cost none/linear; the normalized pattern is inside the compatibility subset | +| `seq.at` | `List[2..19], Int[0..18]` | native | only candidate, cost none/constant | +| `seq.len` | `List[0..2147483647]` | native | only candidate, cost none/constant | +| `seq.len` | `List[2..19]` | native | only candidate, cost none/constant | +| `seq.push` | `` | native | only candidate, cost none/constant | +| `str.asciiLower` | `String[0..2147483647]` | native | native, cost one/linear; a single regex pass maps A-Z and leaves every other scalar alone; rejected `toLowerCase` is only ASCII-equivalent on ASCII input | +| `str.codePoints` | `Digits[2..19] matches ^[0-9]{2,19}$` | native | only candidate, cost one/linear | +| `str.codePoints` | `String[0..2147483647]` | native | only candidate, cost one/linear | +| `str.fromCodePoints` | `List[0..2147483647]` | native | only candidate, cost one/linear | + +Mix: 25 native, 0 library, 0 portable. diff --git a/engine/tests/fixtures/example._driver.ts.txt b/engine/tests/fixtures/example._driver.ts.txt new file mode 100644 index 000000000..22d4cd63f --- /dev/null +++ b/engine/tests/fixtures/example._driver.ts.txt @@ -0,0 +1,27 @@ +// Code generated by the logic engine. DO NOT EDIT. +// source: _driver + +import { createInterface } from "node:readline"; +import { isValidLuhn } from "./is-valid-luhn.ts"; +import { slugify } from "./slugify.ts"; + +type Handler = (args: readonly unknown[], env: unknown) => unknown; + +const handlers: Record = { + "is-valid-luhn::isValidLuhn": (args, env) => isValidLuhn(args[0] as Parameters[0]), + "slugify::slugify": (args, env) => slugify(args[0] as Parameters[0]), +}; + +const reader = createInterface({ input: process.stdin }); + +for await (const line of reader) { + if (line.trim() === "") continue; + const request = JSON.parse(line) as { fn: string; args: unknown[] }; + + try { + const value = await handlers[request.fn]!(request.args, undefined); + process.stdout.write(`${JSON.stringify({ ok: true, value: value === undefined ? null : value })}\n`); + } catch (error) { + process.stdout.write(`${JSON.stringify({ ok: false, error: (error as Error).constructor.name })}\n`); + } +} diff --git a/engine/tests/fixtures/example.core.txt b/engine/tests/fixtures/example.core.txt new file mode 100644 index 000000000..0cd8d8199 --- /dev/null +++ b/engine/tests/fixtures/example.core.txt @@ -0,0 +1,146 @@ +record HttpHeader { name: Ascii[0..2147483647], value: String[0..2147483647] } +record HttpRequest { method: Ascii[3..7], url: String[0..2147483647], headers: List[0..2147483647], body: String[0..2147483647], timeoutMillis: Int[0..600000] } +record HttpResponse { status: Int[0..599], headers: List[0..2147483647], body: String[0..2147483647] } +error HttpError +const is-valid-luhn::DIGITS: String[0..2147483647] +const slugify::HYPHEN: Int[45..45] +const slugify::LOWER_A: Int[97..97] +const slugify::LOWER_Z: Int[122..122] +const slugify::ZERO: Int[48..48] +const slugify::NINE: Int[57..57] + +fn std/date::floorDiv(value: Int[306..3652364]): Int[-1..24] ! Pure + const quotient: Int[0..24] = int.div(value: Int[306..3652364], 146097: Int[146097..146097]): Int[0..24] + if (int.lt(value: Int[306..3652364], 0: Int[0..0]): Bool && !core.eq(int.mul(quotient: Int[0..24], 146097: Int[146097..146097]): Int[0..3506328], value: Int[306..-1]): Bool) { + return int.sub(quotient: Int[0..24], 1: Int[1..1]): Int[-1..23] + } + return quotient: Int[0..24] + +fn std/date::yearFromDays(days: Int[-719162..2932896]): Int[1..9999] ! Pure + const shifted: Int[306..3652364] = int.add(days: Int[-719162..2932896], 719468: Int[719468..719468]): Int[306..3652364] + const era: Int[-1..24] = std/date::floorDiv(shifted: Int[306..3652364]) + const dayOfEra: Int[-3506022..3798461] = int.sub(shifted: Int[306..3652364], int.mul(era: Int[-1..24], 146097: Int[146097..146097]): Int[-146097..3506328]): Int[-3506022..3798461] + const yearOfEra: Int[-9612..10413] = int.div(int.sub(int.add(int.sub(dayOfEra: Int[-3506022..3798461], int.div(dayOfEra: Int[-3506022..3798461], 1460: Int[1460..1460]): Int[-2401..2601]): Int[-3508623..3800862], int.div(dayOfEra: Int[-3506022..3798461], 36524: Int[36524..36524]): Int[-95..103]): Int[-3508718..3800965], int.div(dayOfEra: Int[-3506022..3798461], 146096: Int[146096..146096]): Int[-23..25]): Int[-3508743..3800988], 365: Int[365..365]): Int[-9612..10413] + const year: Int[-10012..20013] = int.add(yearOfEra: Int[-9612..10413], int.mul(era: Int[-1..24], 400: Int[400..400]): Int[-400..9600]): Int[-10012..20013] + const dayOfYear: Int[-7309466..7309348] = int.sub(dayOfEra: Int[-3506022..3798461], int.sub(int.add(int.mul(365: Int[365..365], yearOfEra: Int[-9612..10413]): Int[-3508380..3800745], int.div(yearOfEra: Int[-9612..10413], 4: Int[4..4]): Int[-2403..2603]): Int[-3510783..3803348], int.div(yearOfEra: Int[-9612..10413], 100: Int[100..100]): Int[-96..104]): Int[-3510887..3803444]): Int[-7309466..7309348] + const monthPrime: Int[-238871..238867] = int.div(int.add(int.mul(5: Int[5..5], dayOfYear: Int[-7309466..7309348]): Int[-36547330..36546740], 2: Int[2..2]): Int[-36547328..36546742], 153: Int[153..153]): Int[-238871..238867] + const month: Int[-238868..238858] = (int.lt(monthPrime: Int[-238871..238867], 10: Int[10..10]): Bool ? int.add(monthPrime: Int[-238871..9], 3: Int[3..3]): Int[-238868..12] : int.sub(monthPrime: Int[10..238867], 9: Int[9..9]): Int[1..238858]) + return int.min(int.max((int.le(month: Int[-238868..238858], 2: Int[2..2]): Bool ? int.add(year: Int[-10012..20013], 1: Int[1..1]): Int[-10011..20014] : year: Int[-10012..20013]), 1: Int[1..1]): Int[1..20014], 9999: Int[9999..9999]): Int[1..9999] + +fn std/date::monthFromDays(days: Int[-719162..2932896]): Int[1..12] ! Pure + const shifted: Int[306..3652364] = int.add(days: Int[-719162..2932896], 719468: Int[719468..719468]): Int[306..3652364] + const era: Int[-1..24] = std/date::floorDiv(shifted: Int[306..3652364]) + const dayOfEra: Int[-3506022..3798461] = int.sub(shifted: Int[306..3652364], int.mul(era: Int[-1..24], 146097: Int[146097..146097]): Int[-146097..3506328]): Int[-3506022..3798461] + const yearOfEra: Int[-9612..10413] = int.div(int.sub(int.add(int.sub(dayOfEra: Int[-3506022..3798461], int.div(dayOfEra: Int[-3506022..3798461], 1460: Int[1460..1460]): Int[-2401..2601]): Int[-3508623..3800862], int.div(dayOfEra: Int[-3506022..3798461], 36524: Int[36524..36524]): Int[-95..103]): Int[-3508718..3800965], int.div(dayOfEra: Int[-3506022..3798461], 146096: Int[146096..146096]): Int[-23..25]): Int[-3508743..3800988], 365: Int[365..365]): Int[-9612..10413] + const dayOfYear: Int[-7309466..7309348] = int.sub(dayOfEra: Int[-3506022..3798461], int.sub(int.add(int.mul(365: Int[365..365], yearOfEra: Int[-9612..10413]): Int[-3508380..3800745], int.div(yearOfEra: Int[-9612..10413], 4: Int[4..4]): Int[-2403..2603]): Int[-3510783..3803348], int.div(yearOfEra: Int[-9612..10413], 100: Int[100..100]): Int[-96..104]): Int[-3510887..3803444]): Int[-7309466..7309348] + const monthPrime: Int[-238871..238867] = int.div(int.add(int.mul(5: Int[5..5], dayOfYear: Int[-7309466..7309348]): Int[-36547330..36546740], 2: Int[2..2]): Int[-36547328..36546742], 153: Int[153..153]): Int[-238871..238867] + return int.min(int.max((int.lt(monthPrime: Int[-238871..238867], 10: Int[10..10]): Bool ? int.add(monthPrime: Int[-238871..9], 3: Int[3..3]): Int[-238868..12] : int.sub(monthPrime: Int[10..238867], 9: Int[9..9]): Int[1..238858]), 1: Int[1..1]): Int[1..238858], 12: Int[12..12]): Int[1..12] + +fn std/date::dayFromDays(days: Int[-719162..2932896]): Int[1..31] ! Pure + const shifted: Int[306..3652364] = int.add(days: Int[-719162..2932896], 719468: Int[719468..719468]): Int[306..3652364] + const era: Int[-1..24] = std/date::floorDiv(shifted: Int[306..3652364]) + const dayOfEra: Int[-3506022..3798461] = int.sub(shifted: Int[306..3652364], int.mul(era: Int[-1..24], 146097: Int[146097..146097]): Int[-146097..3506328]): Int[-3506022..3798461] + const yearOfEra: Int[-9612..10413] = int.div(int.sub(int.add(int.sub(dayOfEra: Int[-3506022..3798461], int.div(dayOfEra: Int[-3506022..3798461], 1460: Int[1460..1460]): Int[-2401..2601]): Int[-3508623..3800862], int.div(dayOfEra: Int[-3506022..3798461], 36524: Int[36524..36524]): Int[-95..103]): Int[-3508718..3800965], int.div(dayOfEra: Int[-3506022..3798461], 146096: Int[146096..146096]): Int[-23..25]): Int[-3508743..3800988], 365: Int[365..365]): Int[-9612..10413] + const dayOfYear: Int[-7309466..7309348] = int.sub(dayOfEra: Int[-3506022..3798461], int.sub(int.add(int.mul(365: Int[365..365], yearOfEra: Int[-9612..10413]): Int[-3508380..3800745], int.div(yearOfEra: Int[-9612..10413], 4: Int[4..4]): Int[-2403..2603]): Int[-3510783..3803348], int.div(yearOfEra: Int[-9612..10413], 100: Int[100..100]): Int[-96..104]): Int[-3510887..3803444]): Int[-7309466..7309348] + const monthPrime: Int[-238871..238867] = int.div(int.add(int.mul(5: Int[5..5], dayOfYear: Int[-7309466..7309348]): Int[-36547330..36546740], 2: Int[2..2]): Int[-36547328..36546742], 153: Int[153..153]): Int[-238871..238867] + return int.min(int.max(int.add(int.sub(dayOfYear: Int[-7309466..7309348], int.div(int.add(int.mul(153: Int[153..153], monthPrime: Int[-238871..238867]): Int[-36547263..36546651], 2: Int[2..2]): Int[-36547261..36546653], 5: Int[5..5]): Int[-7309452..7309330]): Int[-14618796..14618800], 1: Int[1..1]): Int[-14618795..14618801], 1: Int[1..1]): Int[1..14618801], 31: Int[31..31]): Int[1..31] + +fn std/date::floorDiv$1(value: Int[0..9999]): Int[-1..24] ! Pure + const quotient: Int[0..24] = int.div(value: Int[0..9999], 400: Int[400..400]): Int[0..24] + if (int.lt(value: Int[0..9999], 0: Int[0..0]): Bool && !core.eq(int.mul(quotient: Int[0..24], 400: Int[400..400]): Int[0..9600], value: Int[0..-1]): Bool) { + return int.sub(quotient: Int[0..24], 1: Int[1..1]): Int[-1..23] + } + return quotient: Int[0..24] + +fn std/date::daysFromCivil(year: Int[1..9999], month: Int[1..12], day: Int[1..31]): Int[-719162..2932896] ! Pure + const shifted: Int[0..9999] = (int.le(month: Int[1..12], 2: Int[2..2]): Bool ? int.sub(year: Int[1..9999], 1: Int[1..1]): Int[0..9998] : year: Int[1..9999]) + const era: Int[-1..24] = std/date::floorDiv$1(shifted: Int[0..9999]) + const yearOfEra: Int[-9600..10399] = int.sub(shifted: Int[0..9999], int.mul(era: Int[-1..24], 400: Int[400..400]): Int[-400..9600]): Int[-9600..10399] + const monthTerm: Int[0..11] = (int.gt(month: Int[1..12], 2: Int[2..2]): Bool ? int.sub(month: Int[3..12], 3: Int[3..3]): Int[0..9] : int.add(month: Int[1..2], 9: Int[9..9]): Int[10..11]) + const dayOfYear: Int[0..367] = int.sub(int.add(int.div(int.add(int.mul(153: Int[153..153], monthTerm: Int[0..11]): Int[0..1683], 2: Int[2..2]): Int[2..1685], 5: Int[5..5]): Int[0..337], day: Int[1..31]): Int[1..368], 1: Int[1..1]): Int[0..367] + const dayOfEra: Int[-3506503..3798697] = int.add(int.sub(int.add(int.mul(yearOfEra: Int[-9600..10399], 365: Int[365..365]): Int[-3504000..3795635], int.div(yearOfEra: Int[-9600..10399], 4: Int[4..4]): Int[-2400..2599]): Int[-3506400..3798234], int.div(yearOfEra: Int[-9600..10399], 100: Int[100..100]): Int[-96..103]): Int[-3506503..3798330], dayOfYear: Int[0..367]): Int[-3506503..3798697] + return int.min(int.max(int.sub(int.add(int.mul(era: Int[-1..24], 146097: Int[146097..146097]): Int[-146097..3506328], dayOfEra: Int[-3506503..3798697]): Int[-3652600..7305025], 719468: Int[719468..719468]): Int[-4372068..6585557], -719162: Int[-719162..-719162]): Int[-719162..6585557], 2932896: Int[2932896..2932896]): Int[-719162..2932896] + +fn std/date::ymdToDays(year: Int[-9007199254740991..9007199254740991], month: Int[-9007199254740991..9007199254740991], day: Int[-9007199254740991..9007199254740991]): Option ! Pure + if (((((int.lt(year: Int[-9007199254740991..9007199254740991], 1: Int[1..1]): Bool || int.gt(year: Int[1..9007199254740991], 9999: Int[9999..9999]): Bool) || int.lt(month: Int[-9007199254740991..9007199254740991], 1: Int[1..1]): Bool) || int.gt(month: Int[1..9007199254740991], 12: Int[12..12]): Bool) || int.lt(day: Int[-9007199254740991..9007199254740991], 1: Int[1..1]): Bool) || int.gt(day: Int[1..9007199254740991], 31: Int[31..31]): Bool) { + return none + } + const days: Int[-719162..2932896] = std/date::daysFromCivil(year: Int[1..9999], month: Int[1..12], day: Int[1..31]) + if ((!core.eq(std/date::yearFromDays(days: Int[-719162..2932896]), year: Int[1..9999]): Bool || !core.eq(std/date::monthFromDays(days: Int[-719162..2932896]), month: Int[1..12]): Bool) || !core.eq(std/date::dayFromDays(days: Int[-719162..2932896]), day: Int[1..31]): Bool) { + return none + } + return some(days: Int[-719162..2932896]) + +fn std/strings::compareScalars(left: String[0..2147483647], right: String[0..2147483647]): Int[-1..1] ! Pure + const leftPoints: List[0..2147483647] = str.codePoints(left: String[0..2147483647]): List[0..2147483647] + const rightPoints: List[0..2147483647] = str.codePoints(right: String[0..2147483647]): List[0..2147483647] + const shared: Int[0..2147483647] = int.min(seq.len(leftPoints: List[0..2147483647]): Int[0..2147483647], seq.len(rightPoints: List[0..2147483647]): Int[0..2147483647]): Int[0..2147483647] + for index: Int[0..2147483646] = 0: Int[0..0] to shared: Int[0..2147483647] { + const a: Int[0..1114111] = opt.orElse(seq.at(leftPoints: List[0..2147483647], index: Int[0..2147483646]): Option, 0: Int[0..0]): Int[0..1114111] + const b: Int[0..1114111] = opt.orElse(seq.at(rightPoints: List[0..2147483647], index: Int[0..2147483646]): Option, 0: Int[0..0]): Int[0..1114111] + if int.lt(a: Int[0..1114111], b: Int[0..1114111]): Bool { + return -1: Int[-1..-1] + } + if int.gt(a: Int[0..1114111], b: Int[0..1114111]): Bool { + return 1: Int[1..1] + } + } + if int.lt(seq.len(leftPoints: List[0..2147483647]): Int[0..2147483647], seq.len(rightPoints: List[0..2147483647]): Int[0..2147483647]): Bool { + return -1: Int[-1..-1] + } + if int.gt(seq.len(leftPoints: List[0..2147483647]): Int[0..2147483647], seq.len(rightPoints: List[0..2147483647]): Int[0..2147483647]): Bool { + return 1: Int[1..1] + } + return 0: Int[0..0] + +fn std/strings::asciiUpperAll(value: String[0..2147483647]): String[0..2147483647] ! Pure + let points: List[0..2147483647] = [] + forEach point: Int[0..1114111] in str.codePoints(value: String[0..2147483647]): List[0..2147483647] { + if (int.ge(point: Int[0..1114111], 97: Int[97..97]): Bool && int.le(point: Int[97..1114111], 122: Int[122..122]): Bool) { + push points <- int.sub(point: Int[97..122], 32: Int[32..32]): Int[65..90] + } else { + push points <- point: Int[0..1114111] + } + } + return str.fromCodePoints(points: List[0..2147483647]): String[0..2147483647] + +fn std/strings::asciiLowerAll(value: String[0..2147483647]): String[0..2147483647] ! Pure + let points: List[0..2147483647] = [] + forEach point: Int[0..1114111] in str.codePoints(value: String[0..2147483647]): List[0..2147483647] { + if (int.ge(point: Int[0..1114111], 65: Int[65..65]): Bool && int.le(point: Int[65..1114111], 90: Int[90..90]): Bool) { + push points <- int.add(point: Int[65..90], 32: Int[32..32]): Int[97..122] + } else { + push points <- point: Int[0..1114111] + } + } + return str.fromCodePoints(points: List[0..2147483647]): String[0..2147483647] + +fn is-valid-luhn::isValidLuhn(value: String[0..2147483647]): Bool ! Pure + if !re.test/^[0-9]{2,19}$/(value: String[0..2147483647]): Bool { + return false: Bool + } + let sum: Int[0..200] = 0: Int[0..0] + const scalars: List[2..19] = str.codePoints(value: Digits[2..19] matches ^[0-9]{2,19}$): List[2..19] + for index: Int[0..18] = 0: Int[0..0] to seq.len(scalars: List[2..19]): Int[2..19] { + const digit: Int[0..9] = int.sub(opt.orElse(seq.at(scalars: List[2..19], index: Int[0..18]): Option, 48: Int[48..48]): Int[48..57], 48: Int[48..48]): Int[0..9] + const doubled: Bool = core.eq(int.mod(int.sub(seq.len(scalars: List[2..19]): Int[2..19], index: Int[0..18]): Int[-16..19], 2: Int[2..2]): Int[-1..1], 0: Int[0..0]): Bool + const weighted: Int[0..18] = (doubled: Bool ? int.mul(digit: Int[0..9], 2: Int[2..2]): Int[0..18] : digit: Int[0..9]) + sum = int.add(sum: Int[0..162], (int.gt(weighted: Int[0..18], 9: Int[9..9]): Bool ? int.sub(weighted: Int[10..18], 9: Int[9..9]): Int[1..9] : weighted: Int[0..9])): Int[0..171] + } + return core.eq(int.mod(sum: Int[0..171], 10: Int[10..10]): Int[0..9], 0: Int[0..0]): Bool + +fn slugify::slugify(title: String[0..2147483647]): Ascii[0..2147483647] ! Pure + let out: List[0..2147483647] = [] + let pendingHyphen: Bool = false: Bool + forEach point: Int[0..1114111] in str.codePoints(str.asciiLower(title: String[0..2147483647]): String[0..2147483647]): List[0..2147483647] { + if ((int.ge(point: Int[0..1114111], 97: Int[97..97]): Bool && int.le(point: Int[97..1114111], 122: Int[122..122]): Bool) || (int.ge(point: Int[0..1114111], 48: Int[48..48]): Bool && int.le(point: Int[48..1114111], 57: Int[57..57]): Bool)) { + if (pendingHyphen: Bool && int.gt(seq.len(out: List[0..2147483647]): Int[0..2147483647], 0: Int[0..0]): Bool) { + push out <- 45: Int[45..45] + } + pendingHyphen = false: Bool + push out <- point: Int[48..122] + } else { + pendingHyphen = true: Bool + } + } + return str.fromCodePoints(out: List[0..2147483647]): Ascii[0..2147483647] diff --git a/engine/tests/fixtures/example.errors.ts.txt b/engine/tests/fixtures/example.errors.ts.txt new file mode 100644 index 000000000..68fde2bf7 --- /dev/null +++ b/engine/tests/fixtures/example.errors.ts.txt @@ -0,0 +1,9 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: errors + +/** The root of every domain error the core raises. */ +export class DomainError extends Error {} + +/** Raised by the engine's own intrinsics. */ +export class HttpError extends DomainError {} diff --git a/engine/tests/fixtures/example.hir.txt b/engine/tests/fixtures/example.hir.txt new file mode 100644 index 000000000..a53db4a19 --- /dev/null +++ b/engine/tests/fixtures/example.hir.txt @@ -0,0 +1,32 @@ +module is-valid-luhn + const DIGITS = /^[0-9]{2,19}$/ + fn isValidLuhn(value: String): Bool + if !re.test(DIGITS, value) + return false + let sum = 0 + const scalars = str.codePoints(value) + for index from 0 to scalars.length + const digit = ((seq.at(scalars, index) ?? 48) - 48) + const doubled = (((scalars.length - index) % 2) === 0) + const weighted = (doubled ? (digit * 2) : digit) + sum += ((weighted > 9) ? (weighted - 9) : weighted) + return ((sum % 10) === 0) + +module slugify + const HYPHEN = 45 + const LOWER_A = 97 + const LOWER_Z = 122 + const ZERO = 48 + const NINE = 57 + fn slugify(title: String): Ascii + let out = [] + let pendingHyphen = false + for point of str.codePoints(str.asciiLower(title)) + if (((point >= LOWER_A) && (point <= LOWER_Z)) || ((point >= ZERO) && (point <= NINE))) + if (pendingHyphen && (out.length > 0)) + out.push(HYPHEN) + pendingHyphen = false + out.push(point) + else + pendingHyphen = true + return str.fromCodePoints(out) diff --git a/engine/tests/fixtures/example.is-valid-luhn.ts.txt b/engine/tests/fixtures/example.is-valid-luhn.ts.txt new file mode 100644 index 000000000..3a5129f59 --- /dev/null +++ b/engine/tests/fixtures/example.is-valid-luhn.ts.txt @@ -0,0 +1,21 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: is-valid-luhn +// content: 1044cd4bd3bd +/** + * Whether a digit string satisfies the Luhn check. + */ +export function isValidLuhn(value: string): boolean { + if (!/^[0-9]{2,19}$/u.test(value)) { + return false; + } + let sum: number = 0; + const scalars: readonly number[] = Array.from(value, (scalar) => scalar.codePointAt(0)!); + for (let index = 0; index < scalars.length; index++) { + const digit: number = ((scalars[index] ?? 48) - 48); + const doubled: boolean = (((scalars.length - index) % 2) === 0); + const weighted: number = (doubled ? (digit * 2) : digit); + sum = (sum + ((weighted > 9) ? (weighted - 9) : weighted)); + } + return ((sum % 10) === 0); +} diff --git a/engine/tests/fixtures/example.slugify.ts.txt b/engine/tests/fixtures/example.slugify.ts.txt new file mode 100644 index 000000000..3c98117c0 --- /dev/null +++ b/engine/tests/fixtures/example.slugify.ts.txt @@ -0,0 +1,23 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: slugify +// content: a8365af608bb +/** + * A lower cased, hyphen separated slug, keeping only ASCII letters and digits. + */ +export function slugify(title: string): string { + let out: number[] = []; + let pendingHyphen: boolean = false; + for (const point of Array.from(title.replace(/[A-Z]/gu, (scalar) => scalar.toLowerCase()), (scalar) => scalar.codePointAt(0)!)) { + if ((((point >= 97) && (point <= 122)) || ((point >= 48) && (point <= 57)))) { + if ((pendingHyphen && (out.length > 0))) { + out.push(45); + } + pendingHyphen = false; + out.push(point); + } else { + pendingHyphen = true; + } + } + return out.map((point) => String.fromCodePoint(point)).join(""); +} diff --git a/engine/tests/fixtures/example.std_strings.ts.txt b/engine/tests/fixtures/example.std_strings.ts.txt new file mode 100644 index 000000000..4a2defcd2 --- /dev/null +++ b/engine/tests/fixtures/example.std_strings.ts.txt @@ -0,0 +1,18 @@ +// Code generated by the logic engine. DO NOT EDIT. +// engine: 0.1.0 +// source: std/strings +// content: f790963d50f6 +/** + * ASCII-only lower casing, for values that are not proven ASCII. + */ +export function asciiLowerAll(value: string): string { + let points: number[] = []; + for (const point of Array.from(value, (scalar) => scalar.codePointAt(0)!)) { + if (((point >= 65) && (point <= 90))) { + points.push((point + 32)); + } else { + points.push(point); + } + } + return points.map((point) => String.fromCodePoint(point)).join(""); +} diff --git a/engine/tests/helpers.ts b/engine/tests/helpers.ts new file mode 100644 index 000000000..22b0c1f16 --- /dev/null +++ b/engine/tests/helpers.ts @@ -0,0 +1,42 @@ +/** Shared helpers for the engine's own tests: compile a source snippet in a temp project. */ + +import { mkdtempSync, mkdirSync, writeFileSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { CompileError } from "../src/diagnostics.ts"; +import type { Diagnostic } from "../src/diagnostics.ts"; +import { compileProject } from "../src/api.ts"; +import type { Compilation, CompileOptions } from "../src/api.ts"; + +export function withProject(files: Record, run: (root: string) => T): T { + const root = mkdtempSync(join(tmpdir(), "logic-engine-")); + try { + for (const [path, source] of Object.entries(files)) { + const full = join(root, path); + mkdirSync(join(full, ".."), { recursive: true }); + writeFileSync(full, source); + } + return run(root); + } finally { + rmSync(root, { recursive: true, force: true }); + } +} + +export function compileSource(files: Record, options: CompileOptions = {}): Compilation { + return withProject(files, (root) => compileProject(root, options)); +} + +/** Compiles a snippet expected to fail, and returns the diagnostics. */ +export function diagnosticsOf(files: Record): readonly Diagnostic[] { + try { + compileSource(files); + } catch (error) { + if (error instanceof CompileError) return error.diagnostics; + throw error; + } + return []; +} + +export function codesOf(files: Record): string[] { + return diagnosticsOf(files).map((diagnostic) => diagnostic.code); +} diff --git a/engine/tests/idioms.spec.ts b/engine/tests/idioms.spec.ts new file mode 100644 index 000000000..a49b4dd5a --- /dev/null +++ b/engine/tests/idioms.spec.ts @@ -0,0 +1,314 @@ +/** + * The idiomatic-TypeScript frontend: for every form the recognizer maps, the idiomatic spelling + * and the namespace spelling must produce the *same* Core — that equivalence is the whole + * correctness argument (docs/semantics.md is updated to describe the subset in these terms), so + * it is asserted here directly rather than testing the two spellings separately. + * + * Forms the recognizer cannot map soundly are rejected instead, in `subset.spec.ts`'s style: each + * test asserts on the diagnostic code, and the block below it is the message an author who does + * not know this engine actually reads. + */ + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { dumpProgram } from "../src/api.ts"; +import { compileSource } from "./helpers.ts"; +import { codesOf } from "./helpers.ts"; + +/** The dumped Core of one function, isolated from the rest of the program (the stdlib, in every case). */ +function coreOf(source: string, fn = "f"): string { + const compilation = compileSource({ "utility.ts": source }); + const dump = dumpProgram(compilation.program); + const marker = `\nfn utility::${fn}(`; + const start = dump.indexOf(marker); + assert.ok(start >= 0, `missing function utility::${fn} in:\n${dump}`); + const rest = dump.slice(start + 1); + const end = rest.indexOf("\n\nfn "); + return (end === -1 ? rest : rest.slice(0, end)).trim(); +} + +/** Asserts that the idiomatic spelling and the namespace spelling check to the identical Core. */ +function sameCore(idiomatic: string, namespaceForm: string, fn = "f"): void { + assert.equal(coreOf(idiomatic, fn), coreOf(namespaceForm, fn)); +} + +function rejects(code: string, body: string): void { + const codes = codesOf({ "utility.ts": body }); + assert.ok(codes.includes(code), `expected ${code}, got ${codes.join(", ") || "no diagnostics"}`); +} + +/* ==================================================================== * + * Strings + * ==================================================================== */ + +test("value.charCodeAt(i) on a proven-ASCII string is str.codeAt", () => { + sameCore( + "export function f(value: AsciiOf<5>, index: IntRange<0, 4>): Int { return value.charCodeAt(index); }", + "export function f(value: AsciiOf<5>, index: IntRange<0, 4>): Int { return str.codeAt(value, index); }", + ); +}); + +test("value.charAt(i) on a proven-ASCII string is str.charAt", () => { + sameCore( + "export function f(value: AsciiOf<5>, index: IntRange<0, 4>): string { return value.charAt(index); }", + "export function f(value: AsciiOf<5>, index: IntRange<0, 4>): string { return str.charAt(value, index); }", + ); +}); + +test("value[i] on a proven-ASCII string with a proven index is also str.charAt", () => { + sameCore( + "export function f(value: AsciiOf<5>, index: IntRange<0, 4>): string { return value[index]; }", + "export function f(value: AsciiOf<5>, index: IntRange<0, 4>): string { return str.charAt(value, index); }", + ); +}); + +test("value[i] with an unprovable index is the checked str.charAtOpt, an Option", () => { + sameCore( + "export function f(value: Ascii, index: Int): string | undefined { return value[index]; }", + "export function f(value: Ascii, index: Int): string | undefined { return str.charAtOpt(value, index); }", + ); +}); + +test("value[i] ?? fallback picks the checked str.charAtOpt even with a proven index", () => { + sameCore( + 'export function f(value: AsciiOf<5>, index: IntRange<0, 4>): string { return value[index] ?? ""; }', + 'export function f(value: AsciiOf<5>, index: IntRange<0, 4>): string { return str.charAtOpt(value, index) ?? ""; }', + ); +}); + +test("value[i]?.charCodeAt(0) is str.codeAtOpt, the checked numeric accessor", () => { + sameCore( + "export function f(value: Ascii, index: Int): Int | undefined { return value[index]?.charCodeAt(0); }", + "export function f(value: Ascii, index: Int): Int | undefined { return str.codeAtOpt(value, index); }", + ); +}); + +test("value[i]?.charCodeAt(0) ?? fallback is str.codeAtOpt ?? fallback, even with a proven index", () => { + sameCore( + "export function f(value: AsciiOf<5>, index: IntRange<0, 4>): Int { return value[index]?.charCodeAt(0) ?? 0; }", + "export function f(value: AsciiOf<5>, index: IntRange<0, 4>): Int { return str.codeAtOpt(value, index) ?? 0; }", + ); +}); + +test("value.slice(a, b) on a proven-ASCII string is str.slice", () => { + sameCore( + "export function f(value: Ascii, a: IntRange<0, 10>, b: IntRange<0, 10>): string { return value.slice(a, b); }", + "export function f(value: Ascii, a: IntRange<0, 10>, b: IntRange<0, 10>): string { return str.slice(value, a, b); }", + ); +}); + +test("value.trim() is str.trim", () => { + sameCore( + "export function f(value: string): string { return value.trim(); }", + "export function f(value: string): string { return str.trim(value); }", + ); +}); + +test("value.padStart(n, c) is str.padStart", () => { + sameCore( + 'export function f(value: string): string { return value.padStart(5, "0"); }', + 'export function f(value: string): string { return str.padStart(value, 5, "0"); }', + ); +}); + +test("value.length on a string is str.len", () => { + sameCore( + "export function f(value: string): Int { return value.length; }", + "export function f(value: string): Int { return str.len(value); }", + ); +}); + +test("String(n) on an Int is str.fromInt", () => { + sameCore( + "export function f(value: Int): string { return String(value); }", + "export function f(value: Int): string { return str.fromInt(value); }", + ); +}); + +test("n.toString() on an Int is str.fromInt", () => { + sameCore( + "export function f(value: Int): string { return value.toString(); }", + "export function f(value: Int): string { return str.fromInt(value); }", + ); +}); + +test("[...s] is str.codePoints", () => { + sameCore( + "export function f(value: string): List { return [...value]; }", + "export function f(value: string): List { return str.codePoints(value); }", + ); +}); + +test('value.replace(/[^0-9]/g, "") is re.retain on the class', () => { + sameCore( + 'export function f(value: string): string { return value.replace(/[^0-9]/g, ""); }', + 'const DIGIT = /^[0-9]$/;\nexport function f(value: string): string { return re.retain(DIGIT, value); }', + ); +}); + +test("value.toUpperCase() on a proven-ASCII string is str.asciiUpper", () => { + sameCore( + "export function f(value: Ascii): Ascii { return value.toUpperCase(); }", + "export function f(value: Ascii): Ascii { return str.asciiUpper(value); }", + ); +}); + +test("value.toLowerCase() on a proven-ASCII string is str.asciiLower", () => { + sameCore( + "export function f(value: Ascii): Ascii { return value.toLowerCase(); }", + "export function f(value: Ascii): Ascii { return str.asciiLower(value); }", + ); +}); + +test("PATTERN.test(value) on a module-level regex constant is re.test", () => { + sameCore( + "const PATTERN = /^[0-9]+$/;\nexport function f(value: string): boolean { return PATTERN.test(value); }", + "const PATTERN = /^[0-9]+$/;\nexport function f(value: string): boolean { return re.test(PATTERN, value); }", + ); +}); + +/* ==================================================================== * + * Lists + * ==================================================================== */ + +test("xs.length on a list is seq.len", () => { + sameCore( + "export function f(xs: List): Int { return xs.length; }", + "export function f(xs: List): Int { return seq.len(xs); }", + ); +}); + +test("xs[i] with a proven index is seq.get", () => { + sameCore( + "export function f(xs: List, index: IntRange<0, 4>): Int { return xs[index]; }", + "export function f(xs: List, index: IntRange<0, 4>): Int { return seq.get(xs, index); }", + ); +}); + +test("xs[i] with an unprovable index is the checked seq.at, an Option", () => { + sameCore( + "export function f(xs: List, index: Int): Int | undefined { return xs[index]; }", + "export function f(xs: List, index: Int): Int | undefined { return seq.at(xs, index); }", + ); +}); + +test("xs[i] ?? fallback picks the checked seq.at even with a proven index", () => { + sameCore( + "export function f(xs: List, index: IntRange<0, 4>): Int { return xs[index] ?? 0; }", + "export function f(xs: List, index: IntRange<0, 4>): Int { return seq.at(xs, index) ?? 0; }", + ); +}); + +/* ==================================================================== * + * Numbers + * ==================================================================== */ + +test("Math.min on Int is int.min", () => { + sameCore( + "export function f(a: Int, b: Int): Int { return Math.min(a, b); }", + "export function f(a: Int, b: Int): Int { return int.min(a, b); }", + ); +}); + +test("Math.max on Int is int.max", () => { + sameCore( + "export function f(a: Int, b: Int): Int { return Math.max(a, b); }", + "export function f(a: Int, b: Int): Int { return int.max(a, b); }", + ); +}); + +test("Math.abs on Int is int.abs", () => { + sameCore( + "export function f(a: Int): Int { return Math.abs(a); }", + "export function f(a: Int): Int { return int.abs(a); }", + ); +}); + +test("Math.trunc on an already-exact Int is the identity", () => { + sameCore( + "export function f(a: Int): Int { return Math.trunc(a); }", + "export function f(a: Int): Int { return a; }", + ); +}); + +test("Math.floor on an already-exact Int is the identity", () => { + sameCore( + "export function f(a: Int): Int { return Math.floor(a); }", + "export function f(a: Int): Int { return a; }", + ); +}); + +/* ==================================================================== * + * Forms with no sound mapping: a diagnostic is the deliverable, not a lowering. + * ==================================================================== */ + +test("charCodeAt on a string that is not proven ASCII is rejected, not silently lowered", () => { + rejects( + "E_UTF16_POSITION", + "export function f(value: string, index: Int): Int { return value.charCodeAt(index); }", + ); +}); + +test("s[i] on a string that is not proven ASCII is rejected", () => { + rejects( + "E_UTF16_POSITION", + "export function f(value: string, index: Int): string | undefined { return value[index]; }", + ); +}); + +test("slice on a string that is not proven ASCII is rejected", () => { + rejects( + "E_UTF16_POSITION", + "export function f(value: string, a: Int, b: Int): string { return value.slice(a, b); }", + ); +}); + +test("toUpperCase on a string that is not proven ASCII is rejected, not silently lowered", () => { + rejects("E_UNICODE_CASE", "export function f(value: string): string { return value.toUpperCase(); }"); +}); + +test("Math.random() is rejected: the Random capability offers only nextU32", () => { + rejects("E_MATH_RANDOM", "export function f(): Float { return Math.random(); }"); +}); + +test("Math.min on Float is rejected: there is no float.min", () => { + rejects( + "E_MATH_FLOAT", + "export function f(a: Float, b: Float): Float { return Math.min(a, b); }", + ); +}); + +test("new Date(...) is rejected: it has no proleptic-Gregorian, non-rolling equivalent", () => { + rejects( + "E_HOST_DATE", + "export function f(y: Int, m: Int, d: Int): CivilDate { return new Date(y, m, d); }", + ); +}); + +test("spreading two lists together is rejected: only [...s] on a single string is admitted", () => { + rejects( + "E_ARRAY_SPREAD", + "export function f(xs: List, ys: List): List { return [...xs, ...ys]; }", + ); +}); + +test(".replace without the g flag is rejected: it would rewrite only the first match", () => { + rejects( + "E_REPLACE_UNSUPPORTED", + 'export function f(value: string): string { return value.replace(/[^0-9]/, ""); }', + ); +}); + +test("value[index]?.charCodeAt(1) is rejected: only the literal 0 is the codeAtOpt idiom", () => { + rejects( + "E_OPTIONAL_CHAIN", + "export function f(value: Ascii, index: Int): Int | undefined { return value[index]?.charCodeAt(1); }", + ); +}); + +test(".replace with a pattern that is not a single class is rejected", () => { + rejects( + "E_REPLACE_UNSUPPORTED", + 'export function f(value: string): string { return value.replace(/[^0-9]{2}/g, ""); }', + ); +}); diff --git a/engine/tests/intrinsics.spec.ts b/engine/tests/intrinsics.spec.ts new file mode 100644 index 000000000..79014a842 --- /dev/null +++ b/engine/tests/intrinsics.spec.ts @@ -0,0 +1,114 @@ +/** + * Intrinsic vectors. + * + * This is the layer of conformance that scales: every intrinsic is checked against its own + * vectors, independently of any utility, so a lowering that disagrees with the specification is + * caught once rather than once per caller. + */ + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { allIntrinsics, lookupIntrinsic } from "../src/intrinsics/index.ts"; +import type { EvalContext } from "../src/intrinsics/index.ts"; +import { NONE, asString, civilDate, decimal, some, valuesEqual } from "../src/values.ts"; +import type { Value } from "../src/values.ts"; + +const CONTEXT: EvalContext = { + http: () => undefined, + now: () => 0n, + sleep: () => undefined, + nextU32: () => 0n, +}; + +function evaluate(name: string, args: Value[]): Value { + const intrinsic = lookupIntrinsic(name); + assert.ok(intrinsic !== undefined, `unknown intrinsic ${name}`); + return intrinsic.evaluate(args, CONTEXT); +} + +function vector(name: string, args: Value[], expected: Value): void { + const actual = evaluate(name, args); + assert.ok( + valuesEqual(actual, expected), + `${name}(${args.map((arg) => String(arg)).join(", ")}) was ${String(actual)}, expected ${String(expected)}`, + ); +} + +test("string operations count scalars, not code units", () => { + vector("str.len", ["café"], 5n); + vector("str.len", ["\u{1f600}"], 1n); + vector("str.codePoints", ["a\u{1f600}"], [97n, 0x1f600n]); + vector("str.fromCodePoints", [[97n, 0x1f600n]], "a\u{1f600}"); +}); + +test("trim removes exactly the 25 code points JavaScript trims", () => { + vector("str.trim", [" a "], "a"); + vector("str.trim", [" a "], "a"); + // U+0085 is whitespace to Python's str.strip but not to JavaScript's trim. + vector("str.trim", ["\u0085a\u0085"], "\u0085a\u0085"); +}); + +test("ASCII case mapping leaves everything else alone", () => { + vector("str.asciiUpper", ["straße"], "STRAßE"); + vector("str.asciiLower", ["STRAßE"], "straße"); +}); + +test("comparison is in scalar order, not UTF-16 order", () => { + // U+1F600 is above U+FFFD as a scalar, but its surrogate pair sorts below in UTF-16. + vector("str.compare", ["\u{1f600}", "�"], 1n); + vector("str.compare", ["a", "b"], -1n); + vector("str.compare", ["ab", "ab"], 0n); +}); + +test("integer division and remainder truncate toward zero", () => { + vector("int.div", [-7n, 2n], -3n); + vector("int.mod", [-7n, 2n], -1n); + vector("int.mod", [7n, -2n], 1n); +}); + +test("decimal arithmetic is exact and names its rounding", () => { + vector("dec.add", [decimal(1234n, 2), decimal(1n, 2)], decimal(1235n, 2)); + vector("dec.mul", [decimal(150n, 2), decimal(3n, 1)], decimal(450n, 3)); + vector("dec.rescale", [decimal(1235n, 3), 2n, "half-even"], decimal(124n, 2)); + vector("dec.rescale", [decimal(1245n, 3), 2n, "half-even"], decimal(124n, 2)); + vector("dec.rescale", [decimal(1245n, 3), 2n, "half-up"], decimal(125n, 2)); + vector("dec.divRound", [decimal(100n, 2), decimal(300n, 2), 4n, "half-up"], decimal(3333n, 4)); +}); + +test("a float converts through its exact binary value", () => { + // 1.005 is really 1.00499999999999989..., which is why half-up gives 1.00 here. + vector("dec.fromFloat", [1.005, 2n, "half-up"], decimal(100n, 2)); + vector("dec.fromFloat", [2.5, 0n, "half-even"], decimal(2n, 0)); + vector("dec.fromFloat", [3.5, 0n, "half-even"], decimal(4n, 0)); +}); + +test("civil dates are exact across the supported range", () => { + vector("date.fromYmd", [2024n, 2n, 29n], some(civilDate(19782))); + vector("date.fromYmd", [2023n, 2n, 29n], NONE); + vector("date.dayOfWeek", [civilDate(0)], 4n); + vector("date.diffDays", [civilDate(10), civilDate(3)], 7n); + vector("date.year", [civilDate(-719162)], 1n); + vector("date.fromEpochDays", [-719163n], NONE); +}); + +test("sorting is stable and compares keys in scalar order", () => { + const items = [ + { __kind: "record" as const, type: "R", fields: { key: 1n, tag: "a" } }, + { __kind: "record" as const, type: "R", fields: { key: 0n, tag: "b" } }, + { __kind: "record" as const, type: "R", fields: { key: 1n, tag: "c" } }, + ]; + const sorted = evaluate("seq.sortStableBy", [ + items, + { __kind: "lambda", call: (args) => (args[0] as (typeof items)[number]).fields["key"]! }, + ]); + assert.deepEqual( + (sorted as typeof items).map((item) => asString(item.fields["tag"]!)), + ["b", "a", "c"], + ); +}); + +test("every intrinsic has documentation", () => { + for (const intrinsic of allIntrinsics()) { + assert.ok(intrinsic.doc.length > 10, `${intrinsic.name} has no documentation`); + } +}); diff --git a/engine/tests/loops.spec.ts b/engine/tests/loops.spec.ts new file mode 100644 index 000000000..9ba969412 --- /dev/null +++ b/engine/tests/loops.spec.ts @@ -0,0 +1,157 @@ +/** + * Loop control flow, which is where the analysis is easiest to get wrong. + * + * `break` and `continue` do not leave the function: an assignment made just before one is live on + * the next iteration, or after the loop. A checker that walks only the path falling out of the + * body would miss that and prove a range the loop can exceed — these tests pin that it does not. + */ + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { typeToString } from "../src/types.ts"; +import { codesOf, compileSource } from "./helpers.ts"; + +function rejects(code: string, body: string): void { + const codes = codesOf({ "utility.ts": body }); + assert.ok(codes.includes(code), `expected ${code}, got ${codes.join(", ") || "no diagnostics"}`); +} + +test("a value assigned before `continue` is live on the next iteration", () => { + rejects( + "E_RETURN_TYPE", + `export function f(items: List): IntRange<0, 0> { + let x: IntRange<0, 19> = 0; + + for (let i = 0; i < items.length; i++) { + if (i === 0) { + x = 15; + continue; + } + } + + return x; +}`, + ); +}); + +test("a value assigned before `break` is live after the loop", () => { + rejects( + "E_RETURN_TYPE", + `export function f(items: List): IntRange<0, 0> { + let x: IntRange<0, 19> = 0; + + for (let i = 0; i < items.length; i++) { + if (i === 0) { + x = 15; + break; + } + } + + return x; +}`, + ); +}); + +test("an index widened on a `continue` path cannot be used unchecked", () => { + rejects( + "E_SIGNATURE", + `export function f(items: List, table: List): Int { + let index: IntRange<0, 19> = 0; + + for (let position = 0; position < items.length; position++) { + if (position === 0) { + index = 15; + continue; + } + + return seq.get(table, index); + } + + return 0; +}`, + ); +}); + +test("the range a loop really holds is the one the checker reports", () => { + const compilation = compileSource({ + "lib/helpers.ts": `export function lastSeen(values: List, 1, 4>): Int { + let seen: IntRange<0, 9> = 0; + + for (const value of values) { + if (value === 3) { + seen = 7; + continue; + } + + seen = value; + } + + return seen; +}`, + "utility.ts": `import { lastSeen } from "./lib/helpers"; + +export function f(values: List, 1, 4>): Int { + return lastSeen(values); +}`, + }); + const helper = compilation.program.functions.get("lib/helpers::lastSeen"); + assert.ok(helper !== undefined); + assert.equal(typeToString(helper.ret), "Int[0..9]"); +}); + +test("a `break` that ends a switch case is dropped rather than translated", () => { + const compilation = compileSource({ + "utility.ts": `export type Kind = "a" | "b"; + +export function f(kind: Kind, values: List, 1, 5>): IntRange<0, 100> { + let total: IntRange<0, 100> = 0; + + for (const value of values) { + switch (kind) { + case "a": + total += value; + break; + default: + total += 1; + } + } + + return total; +}`, + }); + const fn = compilation.program.functions.get("utility::f"); + assert.ok(fn !== undefined); + // A target that prints a switch as a chain of conditionals would read a kept `break` as + // leaving the loop, which is not what the source says. + assert.ok( + !JSON.stringify(fn.body, (_key, value: unknown) => (typeof value === "bigint" ? value.toString() : value)).includes( + '"break"', + ), + "the switch case's break reached the Core", + ); +}); + +test("a `break` that leaves a switch case early is refused", () => { + rejects( + "E_SWITCH_BREAK", + `export type Kind = "a" | "b"; + +export function f(kind: Kind, value: IntRange<0, 9>): Int { + let total: Int = 0; + + switch (kind) { + case "a": + if (value > 3) { + break; + } + + total = 1; + break; + default: + total = 2; + } + + return total; +}`, + ); +}); diff --git a/engine/tests/number-inference.spec.ts b/engine/tests/number-inference.spec.ts new file mode 100644 index 000000000..de8dc4a80 --- /dev/null +++ b/engine/tests/number-inference.spec.ts @@ -0,0 +1,162 @@ +/** + * What the checker does with a bare `number` (docs/subset-gaps.md, "`number` as a parameter or + * return type"). + * + * A parameter's range comes from one of two places — a guard the body writes, for a utility with + * no call site of its own to take it from, or the call sites themselves, for library code, the + * same specialization an explicit `Int` already gets (ADR 0004). Neither is available, the + * refusal is the deliverable: it has to name the parameter, say what could not be proven, and + * name the guard to add, in ordinary code. + */ + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { typeToString } from "../src/types.ts"; +import { compileSource, diagnosticsOf } from "./helpers.ts"; + +function paramType(source: string, fn: string, param: string): string { + const compilation = compileSource({ "utility.ts": source }); + const target = compilation.program.functions.get(`utility::${fn}`); + assert.ok(target !== undefined, `missing ${fn}`); + const binding = target.params.find((item) => item.name === param); + assert.ok(binding !== undefined, `missing parameter ${param}`); + return typeToString(binding.type); +} + +function retType(source: string, fn: string): string { + const compilation = compileSource({ "utility.ts": source }); + const target = compilation.program.functions.get(`utility::${fn}`); + assert.ok(target !== undefined, `missing ${fn}`); + return typeToString(target.ret); +} + +test("a utility's `number` parameter takes its range from an early-exit guard", () => { + const source = `export function clampedInput(n: number): number { + if (n < 0 || n > 99) { + return 0; + } + + return n; +}`; + assert.equal(paramType(source, "clampedInput", "n"), "Int[0..99]"); + assert.equal(retType(source, "clampedInput"), "Int[0..99]"); +}); + +test("a `number` return type is always inferred from the body, no declared ceiling to stay under", () => { + const source = `export function sumOfDigits(value: DigitsOf<3>): number { + return str.codeAt(value, 0) - 48 + (str.codeAt(value, 1) - 48) + (str.codeAt(value, 2) - 48); +}`; + assert.equal(retType(source, "sumOfDigits"), "Int[0..27]"); +}); + +test("a `number` return type is inferred the same way for a parameter that is itself a refinement", () => { + const source = `export function firstDigitValue(value: DigitsOf<11>): number { + return str.codeAt(value, 0) - 48; +}`; + assert.equal(retType(source, "firstDigitValue"), "Int[0..9]"); +}); + +test("a library helper's `number` parameter is specialized per call site, exactly like Int", () => { + const source = `function digitAt(value: Digits, index: number): number { + return value.charCodeAt(index) - 48; +} + +export function f(value: DigitsOf<4>): IntRange<0, 9> { + return digitAt(value, 3); +}`; + // Unoptimized, because this is about what the checker proved, and the optimizer acts on that + // proof: an `Int[3..3]` parameter is folded to the constant at every read and then dropped from + // the signature entirely. The test below covers that half. + const compilation = compileSource({ "utility.ts": source }, { noOptimize: true }); + const specialized = compilation.program.functions.get("utility::digitAt"); + assert.ok(specialized !== undefined, "the helper was not specialized"); + assert.equal(typeToString(specialized.params[0]!.type), "Digits[4]"); + // The index came from the call site (a proven literal 3), not the wide default `number` starts as. + assert.equal(typeToString(specialized.params[1]!.type), "Int[3..3]"); + assert.equal(typeToString(specialized.ret), "Int[0..9]"); +}); + +test("a parameter proven to be one integer is folded to it and then dropped from the signature", () => { + const source = `function digitAt(value: Digits, index: number): number { + return value.charCodeAt(index) - 48; +} + +export function f(value: DigitsOf<4>): IntRange<0, 9> { + return digitAt(value, 3); +}`; + const specialized = compileSource({ "utility.ts": source }).program.functions.get("utility::digitAt"); + assert.ok(specialized !== undefined, "the helper was not specialized"); + // `index` is gone: every read of it became `3`, which left nothing for the parameter to carry. + // Go and Rust both refuse to compile an unused parameter, so this is correctness, not tidiness. + assert.deepEqual( + specialized.params.map((param) => param.name), + ["value"], + ); +}); + +test("an exported `number` parameter no guard narrows is refused, naming the guard to add", () => { + const diagnostics = diagnosticsOf({ + "utility.ts": `export function isPositive(n: number): boolean { + return n > 0; +}`, + }); + assert.equal(diagnostics.length, 1); + const [diagnostic] = diagnostics; + assert.equal(diagnostic!.code, "E_BARE_NUMBER"); + assert.equal( + diagnostic!.message, + "`n` is a bare `number`, and `isPositive` never narrows it before using it", + ); + assert.equal( + diagnostic!.suggestion, + "add a guard before n is used, for example `if (n < 0 || n > 99) return …;` — the range a " + + "guard like that proves becomes n's published contract; write `n: Int` instead if the full " + + "range really is what is meant", + ); +}); + +test("an exported `number` parameter that is never used at all is refused, not silently accepted", () => { + const diagnostics = diagnosticsOf({ + "utility.ts": `export function alwaysTrue(n: number): boolean { + return true; +}`, + }); + assert.equal(diagnostics.length, 1); + const [diagnostic] = diagnostics; + assert.equal(diagnostic!.code, "E_BARE_NUMBER"); + assert.equal(diagnostic!.message, "`n` is never used, so `alwaysTrue` proves nothing about its range"); + assert.equal( + diagnostic!.suggestion, + "remove the parameter, or write `n: Int` if the full range really is what is meant", + ); +}); + +test("reading a `number` parameter both inside and outside a guard still refuses", () => { + // `n` is read once inside the guard (excluded — the guard's own test proves nothing about a use + // that has not happened yet) and once for real before the guard runs, so nothing narrows the + // value at the point it is actually used. + const diagnostics = diagnosticsOf({ + "utility.ts": `export function unclamped(n: number): number { + const raw = n; + + if (n < 0 || n > 99) { + return 0; + } + + return raw; +}`, + }); + assert.equal(diagnostics.length, 1); + assert.equal(diagnostics[0]!.code, "E_BARE_NUMBER"); +}); + +test("an explicit `Int` still opts out of inference and needs no guard", () => { + const compilation = compileSource({ + "utility.ts": `export function identity(n: Int): Int { + return n; +}`, + }); + const target = compilation.program.functions.get("utility::identity"); + assert.ok(target !== undefined); + assert.equal(typeToString(target.params[0]!.type), "Int[-9007199254740991..9007199254740991]"); +}); diff --git a/engine/tests/refinements.spec.ts b/engine/tests/refinements.spec.ts new file mode 100644 index 000000000..e9b4d6cce --- /dev/null +++ b/engine/tests/refinements.spec.ts @@ -0,0 +1,151 @@ +/** + * What the checker must *accept*: the refinements that make a native lowering safe. + * + * Each case pins the inferred type, because the range and the character class are the proof a + * backend reads when it picks a representation or a native call. + */ + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { typeToString } from "../src/types.ts"; +import { compileSource } from "./helpers.ts"; + +/** + * The inferred result of a library helper. + * + * A utility keeps its declared signature, because that signature is the published contract; a + * helper is specialized per call site, so its type is what the analysis actually proved. + */ +function helperType(source: string, fn: string, entry: string): string { + const compilation = compileSource({ "lib/helpers.ts": source, "utility.ts": entry }); + const target = compilation.program.functions.get(`lib/helpers::${fn}`); + assert.ok(target !== undefined, `missing ${fn}`); + return typeToString(target.ret); +} + +/** A one line utility that calls the helper under test, so the helper is specialized. */ +const ENTRY = [ + 'import { HELPER } from "./lib/helpers";', + "", + "export function f(ARGUMENT): Int {", + "\treturn HELPER(value);", + "}", + "", +].join("\n"); + +function compiles(files: Record): void { + assert.doesNotThrow(() => compileSource(files)); +} + +test("a regex guard refines the class and the length of a string", () => { + const source = `const PATTERN = /^[0-9]{11}$/; + +export function digitOf(value: string): Int { + if (!re.test(PATTERN, value)) { + return -1; + } + + // Provable only because the guard proved 11 digits: the class and the length both come from + // the pattern itself. + return str.codeAt(value, 10) - 48; +}`; + assert.equal( + helperType( + source, + "digitOf", + ENTRY.replaceAll("HELPER", "digitOf").replace("ARGUMENT", "value: string"), + ), + "Int[-1..9]", + ); +}); + +test("a length check narrows a string so indexing is provable", () => { + const source = `export function lastDigit(value: Digits): Int { + if (value.length !== 3) { + return -1; + } + + return str.codeAt(value, 2) - 48; +}`; + assert.equal( + helperType( + source, + "lastDigit", + ENTRY.replaceAll("HELPER", "lastDigit").replace("ARGUMENT", "value: Digits"), + ), + "Int[-1..9]", + ); +}); + +test("a loop proves the exact range of an accumulator", () => { + const source = `export function sumOf(value: DigitsOf<3>): IntRange<0, 27> { + let total: IntRange<0, 27> = 0; + + for (let index = 0; index < 3; index++) { + total += str.codeAt(value, index) - 48; + } + + return total; +}`; + assert.equal( + helperType(source, "sumOf", ENTRY.replaceAll("HELPER", "sumOf").replace("ARGUMENT", "value: DigitsOf<3>")), + "Int[0..27]", + ); +}); + +test("a comparison narrows an integer in both branches", () => { + const source = `export function clamped(value: IntRange<0, 20>): IntRange<0, 9> { + return value < 2 ? 0 : (value > 10 ? 9 : value - 2); +}`; + assert.equal( + helperType(source, "clamped", ENTRY.replaceAll("HELPER", "clamped").replace("ARGUMENT", "value: IntRange<0, 20>")), + "Int[0..9]", + ); +}); + +test("an Option is narrowed by an explicit undefined check", () => { + compiles({ + "utility.ts": `export function f(value: string): Int { + const digits = str.asDigits(value); + + if (digits === undefined) { + return -1; + } + + return digits.length; +}`, + }); +}); + +test("a refinement crosses a function boundary through specialization", () => { + const source = `function digitAt(value: Digits, index: Int): IntRange<0, 9> { + return str.codeAt(value, index) - 48; +} + +export function f(value: DigitsOf<4>): IntRange<0, 9> { + return digitAt(value, 3); +}`; + const compilation = compileSource({ "utility.ts": source }); + const specialized = compilation.program.functions.get("utility::digitAt"); + assert.ok(specialized !== undefined, "the helper was not specialized"); + assert.equal(typeToString(specialized.params[0]!.type), "Digits[4]"); +}); + +test("a string that is only ASCII after a checked conversion may be indexed", () => { + compiles({ + "utility.ts": `export function f(value: string): Int { + const ascii = str.asAscii(value); + + if (ascii === undefined) { + return -1; + } + + if (ascii.length === 0) { + return -2; + } + + // Both facts are needed here: ASCII from the conversion, non-empty from the length check. + return str.codeAt(ascii, 0); +}`, + }); +}); diff --git a/engine/tests/regex.spec.ts b/engine/tests/regex.spec.ts new file mode 100644 index 000000000..e610865f8 --- /dev/null +++ b/engine/tests/regex.spec.ts @@ -0,0 +1,72 @@ +/** + * The regex subset: what it accepts, what it refuses, and that a normalized pattern means the + * same thing as the JavaScript literal it came from. + */ + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { RegexError, normalizeRegex, printRegex, regexMatches } from "../src/regex.ts"; +import { codePointsOf } from "../src/values.ts"; + +const PATTERNS = [ + "^[0-9]{3}$", + "^[0-9]{3}[ .\\-/]*[0-9]{2}$", + "^(?:abc|de)+$", + "^[A-Z][0-9]?$", + "^[^0-9]+$", + "^[0-9]{2,4}$", + "^a*b+c?$", +]; + +const INPUTS = ["", "123", "12", "abc", "abcde", "de", "A1", "A", "aaabbc", "123.45", "1234", "x"]; + +test("a normalized pattern matches what the JavaScript literal matches", () => { + for (const pattern of PATTERNS) { + const normalized = normalizeRegex(pattern); + const native = new RegExp(pattern, "u"); + for (const input of INPUTS) { + assert.equal( + regexMatches(normalized, codePointsOf(input)), + native.test(input), + `${pattern} disagreed on ${JSON.stringify(input)}`, + ); + } + } +}); + +test("a printed pattern still matches what the original did", () => { + for (const pattern of PATTERNS) { + const printed = printRegex(normalizeRegex(pattern).node, "javascript"); + const native = new RegExp(`^${printed}$`, "u"); + const original = new RegExp(pattern, "u"); + for (const input of INPUTS) { + assert.equal(native.test(input), original.test(input), `${printed} disagreed on ${JSON.stringify(input)}`); + } + } +}); + +test("RE2 gets code points in its own syntax", () => { + const printed = printRegex(normalizeRegex("^[\\u00e0-\\u00ff]$").node, "go"); + assert.ok(printed.includes("\\x{e0}"), printed); + assert.ok(!printed.includes("\\u"), printed); +}); + +test("the shorthand classes are refused", () => { + for (const pattern of ["^\\d+$", "^\\w+$", "^\\s+$", "^a\\b$"]) { + assert.throws(() => normalizeRegex(pattern), RegexError, pattern); + } +}); + +test("lookaround, laziness and unanchored patterns are refused", () => { + for (const pattern of ["^(?=a)a$", "^a*?$", "[0-9]+", "^.$"]) { + assert.throws(() => normalizeRegex(pattern), RegexError, pattern); + } +}); + +test("a digits-only pattern is recognized as a refinement", () => { + const normalized = normalizeRegex("^[0-9]{11}$"); + assert.equal(normalized.digitsOnly, true); + assert.equal(normalized.asciiOnly, true); + assert.equal(normalized.minLength, 11); + assert.equal(normalized.maxLength, 11); +}); diff --git a/engine/tests/rust-regex.spec.ts b/engine/tests/rust-regex.spec.ts new file mode 100644 index 000000000..46bc5ef27 --- /dev/null +++ b/engine/tests/rust-regex.spec.ts @@ -0,0 +1,198 @@ +/** + * The Rust target's regex lowering, checked against `RegExp` through the actual compiled crate. + * + * `regex.spec.ts` checks the frontend's own reference matcher (`regexMatches`) against `RegExp`; + * this file checks a different, later link in the same chain: that the Rust code the engine + * generates for `re.test` -- a dedicated scanner for a "chain" pattern, a backtracking matcher for + * everything else (see `engine/src/targets/rust/index.ts`'s "Regex" section) -- decides the same + * thing `RegExp` does, for patterns exercising both paths and the boundary between them. Nothing + * in `core/source` needs the fallback path today, so nothing else in this repository compiles and + * runs it; this is the one place that does. + * + * It builds a small synthetic project (not committed, in the OS temp directory), generates the + * Rust target for it, and runs the generated differential driver -- the same JSON-lines-over- + * stdin protocol `core/conformance/run.ts` uses -- so this is an end-to-end check of the compiled + * binary, not a check of the TypeScript generator's output as text. + */ + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { dirname, join } from "node:path"; +import { tmpdir } from "node:os"; +import { compileProject } from "../src/api.ts"; +import { generate } from "../src/backend/generate.ts"; +import { RUST_BACKEND } from "../src/targets/rust/index.ts"; +import { runTarget } from "../src/conformance/differential.ts"; +import type { Case } from "../src/conformance/differential.ts"; + +/** + * One pattern under test: a source-language regex literal, the exported function name given to + * it, and the inputs to check it against `RegExp` on. `shape` is not read by the test itself -- + * it documents, next to each pattern, which of `chainElementsOf`'s branches is expected to take + * it, so a future reader can tell a real regression from a classifier that just got pickier. + */ +type PatternCase = { + readonly fn: string; + readonly pattern: string; + readonly shape: "scanner" | "fallback"; + readonly inputs: readonly string[]; +}; + +const CASES: readonly PatternCase[] = [ + { + fn: "cpfLike", + pattern: "/^[0-9]{3}[ .\\-/]*[0-9]{2}$/", + shape: "scanner", // fixed digits, then a separator run disjoint from digits, alternating. + inputs: ["123.45", "12345", "123..45", "12a45", "1234", "", "123 45", "123-45", "12345 ", "abc"], + }, + { + fn: "varTail", + pattern: "/^[0-9]{2}[a-z]*$/", + shape: "scanner", // the variable run is last: nothing after it to disagree with. + inputs: ["12", "12abc", "1", "12ABC", "123abc", "ab12", "99z"], + }, + { + fn: "adjacentVarVar", + pattern: "/^a*b+c?$/", + // Two variable-length runs back to back: `chainElementsOf` requires a fixed run between any + // two variable ones (see its comment), so this always falls back, even though `a`/`b`/`c` + // happen to be disjoint here and a smarter scanner could in principle take it. + shape: "fallback", + inputs: ["", "a", "b", "ab", "aabbb", "aabbbc", "abc", "aabbc", "c", "bc", "aabbbcc", "abcabc"], + }, + { + fn: "altGroup", + pattern: "/^(?:abc|de)+$/", + shape: "fallback", // alternation. + inputs: ["abc", "de", "abcde", "deabc", "abcabc", "dede", "ab", "abcd", "", "abcdeabc"], + }, + { + fn: "overlapBacktrack", + pattern: "/^[a-z]*[a-c]{2}$/", + // The variable run's class (`a`-`z`) is not disjoint from what follows it (`a`-`c`), so the + // maximal-munch argument does not hold: a real match can need the `*` to give back + // characters it greedily took ("aaac" only matches by backing off to "aa" + "ac"). This is + // the case a wrong scanner would get wrong, which is exactly why `chainElementsOf` refuses + // it rather than emitting one. + shape: "fallback", + inputs: ["aaac", "aac", "ac", "aaaac", "zzac", "zzzzac", "aa", "a", "", "abac", "zzzzzz", "aaa"], + }, + { + fn: "repeatedGroup", + pattern: "/^(?:ab){2,3}$/", + shape: "fallback", // a repeated group, not a repeated single class. + inputs: ["abab", "ababab", "ab", "abababab", "aba", "", "abx"], + }, + { + fn: "singleClass", + pattern: "/^[0-9]$/", + shape: "scanner", // a bare class, no `Seq` wrapper at all. + inputs: ["5", "55", "", "a"], + }, + { + fn: "emptyPattern", + pattern: "/^$/", + shape: "scanner", // the empty chain: matches only the empty string. + inputs: ["", "a", " "], + }, + { + fn: "boundedVar", + pattern: "/^[0-9]{2,4}$/", + shape: "scanner", // variable but last, and its bound is finite rather than unbounded. + inputs: ["12", "123", "1234", "1", "12345", ""], + }, + { + fn: "twoFixedAdjacent", + pattern: "/^[0-9]{2}[a-z]{3}$/", + shape: "scanner", // two fixed-count runs back to back: no ambiguity to check for. + inputs: ["12abc", "12ab", "123abc", "12abcd", "12ABC"], + }, +]; + +function buildProject(): string { + const root = mkdtempSync(join(tmpdir(), "engine-rust-regex-")); + const source = join(root, "source"); + mkdirSync(source, { recursive: true }); + const lines = CASES.flatMap((entry) => [ + `const ${entry.fn.toUpperCase()} = ${entry.pattern};`, + `export function ${entry.fn}(value: string): boolean {`, + `\treturn re.test(${entry.fn.toUpperCase()}, value);`, + "}", + "", + ]); + writeFileSync(join(source, "regex-cases.ts"), lines.join("\n")); + return root; +} + +test("the compiled Rust driver agrees with RegExp on every case, scanner and fallback alike", () => { + const root = buildProject(); + try { + const compilation = compileProject(join(root, "source")); + const result = generate(compilation.program, RUST_BACKEND); + + const outDir = join(root, "out"); + for (const file of result.files) { + const path = join(outDir, file.path); + mkdirSync(dirname(path), { recursive: true }); + writeFileSync(path, file.text); + } + + // Confirms the classifier put each pattern where this file says it should have, before + // trusting the compiled answers below -- a pattern silently changing shape is exactly the + // kind of regression `docs/targets/rust.md`'s rule is supposed to make visible. + const supportText = result.files.find((file) => file.path === "src/support.rs")?.text ?? ""; + const moduleText = result.files.find((file) => file.path === "src/regex_cases.rs")?.text ?? ""; + for (const entry of CASES) { + const called = moduleText.includes(`re_match_`) && moduleTextCalls(moduleText, entry.fn); + assert.equal( + called, + entry.shape === "scanner", + `${entry.fn} (${entry.pattern}) expected a ${entry.shape} lowering`, + ); + } + if (CASES.some((entry) => entry.shape === "fallback")) { + assert.ok(supportText.includes("fn re_test("), "a fallback pattern exists but re_test was not emitted"); + } + + const cases: Case[] = CASES.flatMap((entry) => + entry.inputs.map((input) => ({ fn: `regex-cases::${entry.fn}`, args: [input] })), + ); + const outcomes = runTarget( + { + name: "rust", + command: "cargo", + args: ["run", "--offline", "--quiet", "--bin", "driver"], + cwd: outDir, + }, + cases, + ); + + let index = 0; + for (const entry of CASES) { + const native = new RegExp(entry.pattern.slice(1, entry.pattern.lastIndexOf("/")), "u"); + for (const input of entry.inputs) { + const outcome = outcomes[index]!; + assert.ok(outcome.ok, `${entry.fn}(${JSON.stringify(input)}) failed: ${JSON.stringify(outcome)}`); + assert.equal( + (outcome as { ok: true; value: unknown }).value, + native.test(input), + `${entry.fn} (${entry.pattern}) disagreed with RegExp on ${JSON.stringify(input)}`, + ); + index++; + } + } + } finally { + rmSync(root, { recursive: true, force: true }); + } +}); + +/** Whether `moduleText`'s definition of `fnName` calls a generated `re_match_N` scanner. */ +function moduleTextCalls(moduleText: string, fnName: string): boolean { + const snake = fnName.replace(/[A-Z]/g, (letter) => `_${letter.toLowerCase()}`); + const start = moduleText.indexOf(`fn ${snake}(`); + if (start < 0) return false; + const end = moduleText.indexOf("\n}", start); + const body = moduleText.slice(start, end < 0 ? undefined : end); + return /re_match_\d+\(/.test(body); +} diff --git a/engine/tests/size.spec.ts b/engine/tests/size.spec.ts new file mode 100644 index 000000000..0785c13be --- /dev/null +++ b/engine/tests/size.spec.ts @@ -0,0 +1,114 @@ +/** + * What the TypeScript target's inlining is allowed to cost. + * + * The npm package this engine generates for is tree-shakeable, and ADR 0012 records that as a + * requirement: a consumer who imports one utility must not carry another. So for that one target + * inlining is not a free win — every copy of a callee is bytes over the wire — and the budget + * that bounds it is asserted here rather than remembered. `engine/scripts/size.ts` measures the + * result the way a consumer's bundler would; these tests cover the decisions behind it. + */ + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { dumpProgram } from "../src/api.ts"; +import { generate } from "../src/backend/generate.ts"; +import { inlineCalls } from "../src/optimize/inline.ts"; +import { TYPESCRIPT_BACKEND } from "../src/targets/typescript/index.ts"; +import { PYTHON_BACKEND } from "../src/targets/python/index.ts"; +import { compileSource } from "./helpers.ts"; + +/** The generated TypeScript of one file, by its path in the output. */ +function generated(source: string, path: string, backend = TYPESCRIPT_BACKEND): string { + const result = generate(compileSource({ "utility.ts": source }).program, backend); + const file = result.files.find((candidate) => candidate.path === path); + assert.ok(file !== undefined, `no ${path} in ${result.files.map((each) => each.path).join(", ")}`); + return file.text; +} + +const SMALL_HELPER_MANY_CALLS = `function digitAt(value: Digits, index: Int): IntRange<0, 9> { + return value.charCodeAt(index) - 48; +} + +export function f(value: DigitsOf<4>): IntRange<0, 36> { + return digitAt(value, 0) + digitAt(value, 1) + digitAt(value, 2) + digitAt(value, 3); +}`; + +test("a one-expression helper is substituted as an expression, not through a synthetic Option", () => { + const text = generated(SMALL_HELPER_MANY_CALLS, "utility.ts"); + assert.match(text, /charCodeAt/u, "the helper's body did not reach the call site"); + assert.doesNotMatch(text, /Result/u, "the splice went through the early-return sentinel"); + assert.doesNotMatch(text, /const _inl\d+Value/u, "a literal or local argument was bound instead of substituted"); +}); + +const BIG_HELPER_MANY_CALLS = `function classify(value: Digits, index: Int): IntRange<0, 4> { + const digit = value.charCodeAt(index) - 48; + + if (digit === 0) { + return 0; + } + + if (digit < 3) { + return 1; + } + + if (digit < 6) { + return 2; + } + + if (digit < 9) { + return 3; + } + + return 4; +} + +export function f(value: DigitsOf<4>): IntRange<0, 16> { + return classify(value, 0) + classify(value, 1) + classify(value, 2) + classify(value, 3); +}`; + +test("a helper too big to duplicate is left as a call in TypeScript and inlined in Python", () => { + // The same program, the same pass, two answers, because the two targets pay for it in different + // currencies: bytes downloaded on one side, interpreter frames on the other. That divergence is + // the point of a per-target budget, so it is asserted directly. + assert.match(generated(BIG_HELPER_MANY_CALLS, "utility.ts"), /\bclassify\(/u, "TypeScript duplicated a large helper"); + assert.doesNotMatch( + generated(BIG_HELPER_MANY_CALLS, "utility.py", PYTHON_BACKEND), + /\bclassify\(/u, + "Python left a call it had the budget to inline", + ); +}); + +const STRAIGHT_LINE_HELPER = `function weigh(value: Digits, index: Int): IntRange<0, 90> { + const digit = value.charCodeAt(index) - 48; + const doubled = digit * 2; + const shifted = doubled * 5; + + return shifted; +} + +`; + +test("a straight-line helper with one call site is absorbed, and the same helper shared is not", () => { + const once = `${STRAIGHT_LINE_HELPER}export function f(value: DigitsOf<4>): IntRange<0, 90> { + return weigh(value, 0); +}`; + const text = generated(once, "utility.ts"); + assert.doesNotMatch(text, /\bweigh\(/u, "the only call site was not absorbed"); + assert.doesNotMatch(text, /function weigh/u, "the absorbed helper was still emitted"); + + // The same helper, called twice, is a copy the budget will not pay for — the difference is the + // call count, not the helper. Both calls pass the same index on purpose: a different constant + // would specialize into a second function (ADR 0004), each with a call site of its own, and + // then there would be nothing shared to refuse. + const twice = `${STRAIGHT_LINE_HELPER}export function f(left: DigitsOf<4>, right: DigitsOf<4>): IntRange<0, 180> { + return weigh(left, 0) + weigh(right, 0); +}`; + assert.match(generated(twice, "utility.ts"), /\bweigh\(/u, "a duplicated helper was inlined anyway"); +}); + +test("a budget of zero duplicated nodes still absorbs a sole call site", () => { + const program = compileSource({ "utility.ts": SMALL_HELPER_MANY_CALLS }).program; + const unchanged = inlineCalls(program, { maxStatements: 8, maxDuplicatedNodes: 0 }); + // Four call sites, so no copy pays for itself and the helper stays. + assert.match(dumpProgram(unchanged), /utility::digitAt/u); +}); diff --git a/engine/tests/snapshots.spec.ts b/engine/tests/snapshots.spec.ts new file mode 100644 index 000000000..ad3ec9cde --- /dev/null +++ b/engine/tests/snapshots.spec.ts @@ -0,0 +1,49 @@ +/** + * Stage snapshots. + * + * Every stage has a readable dump, and the dumps are committed: a change in the HIR, in the + * annotated Core or in the generated TypeScript shows up as a reviewable diff rather than as a + * surprise in a target. Refresh them with `UPDATE_SNAPSHOTS=1 npm test`. + */ + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; +import { compileProject, dumpHir, dumpProgram } from "../src/api.ts"; +import { generate } from "../src/backend/generate.ts"; +import { TYPESCRIPT_BACKEND } from "../src/targets/typescript/index.ts"; + +const EXAMPLE = join(import.meta.dirname, "..", "examples", "generic", "source"); +const FIXTURES = join(import.meta.dirname, "fixtures"); + +function snapshot(name: string, actual: string): void { + const path = join(FIXTURES, name); + mkdirSync(FIXTURES, { recursive: true }); + if (process.env["UPDATE_SNAPSHOTS"] === "1" || !existsSync(path)) { + writeFileSync(path, actual); + return; + } + assert.equal(actual, readFileSync(path, "utf8"), `${name} changed; review the diff and refresh if intended`); +} + +test("the HIR dump is stable", () => { + const compilation = compileProject(EXAMPLE); + const dumps = compilation.modules + .filter((module) => !module.path.startsWith("std/")) + .map((module) => dumpHir(module)) + .join("\n\n"); + snapshot("example.hir.txt", `${dumps}\n`); +}); + +test("the annotated Core dump is stable", () => { + snapshot("example.core.txt", `${dumpProgram(compileProject(EXAMPLE).program)}\n`); +}); + +test("the generated TypeScript is stable", () => { + const result = generate(compileProject(EXAMPLE).program, TYPESCRIPT_BACKEND); + for (const file of result.files) { + snapshot(`example.${file.path.replaceAll("/", "_")}.txt`, file.text); + } + snapshot("example.LOWERING.md", result.lowering); +}); diff --git a/engine/tests/subset.spec.ts b/engine/tests/subset.spec.ts new file mode 100644 index 000000000..56ac4e492 --- /dev/null +++ b/engine/tests/subset.spec.ts @@ -0,0 +1,157 @@ +/** + * Every construct the subset rejects has a test, and every test asserts on the diagnostic code so + * the message can be rewritten without breaking the suite. + */ + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { codesOf } from "./helpers.ts"; + +function rejects(code: string, body: string): void { + const codes = codesOf({ "utility.ts": body }); + assert.ok(codes.includes(code), `expected ${code}, got ${codes.join(", ") || "no diagnostics"}`); +} + +test("a bare number no guard narrows is rejected", () => { + // `number` itself is ordinary TypeScript, accepted and inferred (tests/number-inference.spec.ts) + // — what still rejects is a utility whose body never proves a range for one, which is this case: + // `value` is only ever read at its unconstrained default. + rejects("E_BARE_NUMBER", "export function f(value: number): boolean { return value > 0; }"); +}); + +test("any is rejected", () => { + rejects("E_ANY", "export function f(value: any): boolean { return true; }"); +}); + +test("null is rejected", () => { + rejects("E_NULL", "export function f(value: string): boolean { return value === null; }"); +}); + +test("loose equality is rejected", () => { + rejects("E_LOOSE_EQUALITY", "export function f(value: Int): boolean { return value == 1; }"); +}); + +test("while loops are rejected", () => { + rejects( + "E_WHILE", + "export function f(value: Int): boolean { while (value > 0) { value = value - 1; } return true; }", + ); +}); + +test("try/catch is rejected", () => { + rejects("E_TRY", "export function f(value: Int): boolean { try { return true; } catch { return false; } }"); +}); + +test("host globals are rejected", () => { + // `Math.random()` and the other named `Math` members each get their own, more specific + // diagnostic (`idioms.spec.ts`); `Math.PI` is not one of those, so it keeps the generic one. + rejects("E_HOST_GLOBAL", "export function f(): Float { return Math.PI; }"); +}); + +test("truthiness is rejected", () => { + rejects("E_TRUTHINESS", "export function f(value: Int): boolean { if (value) { return true; } return false; }"); +}); + +test("an unproven index is rejected", () => { + rejects( + "E_SIGNATURE", + "export function f(value: Ascii, index: Int): Int { return str.codeAt(value, index); }", + ); +}); + +test("a possibly zero divisor is rejected", () => { + rejects("E_SIGNATURE", "export function f(left: Int, right: Int): Int { return left / right; }"); +}); + +test("a string that is not proven ASCII cannot be indexed", () => { + rejects("E_SIGNATURE", "export function f(value: string): Int { return str.codeAt(value, 0); }"); +}); + +test("recursion is rejected", () => { + rejects( + "E_RECURSION", + "export function f(value: IntRange<0, 10>): Int { return value === 0 ? 0 : f(value - 1); }", + ); +}); + +test("an unannotated return type is rejected", () => { + rejects("E_MISSING_RETURN_TYPE", "export function f(value: Int) { return value; }"); +}); + +test("a regex with \\d is rejected", () => { + rejects( + "E_REGEX", + "const PATTERN = /^\\d+$/;\nexport function f(value: string): boolean { return re.test(PATTERN, value); }", + ); +}); + +test("an unanchored regex is rejected", () => { + rejects( + "E_REGEX", + "const PATTERN = /[0-9]+/;\nexport function f(value: string): boolean { return re.test(PATTERN, value); }", + ); +}); + +test("a non-exhaustive switch is rejected", () => { + rejects( + "E_NON_EXHAUSTIVE", + `export type Version = "1" | "2"; +export function f(value: Version): Int { + switch (value) { + case "1": + return 1; + } + return 0; +}`, + ); +}); + +test("mutating a non-local is rejected", () => { + rejects( + "E_ASSIGN_TARGET", + `export type Point = { x: Int }; +export function f(point: Point): Int { point.x = 1; return point.x; }`, + ); +}); + +test("interpolating a non-string is rejected", () => { + rejects("E_INTERPOLATION", "export function f(value: Int): string { return `${value}`; }"); +}); + +test("a float compared with === is rejected", () => { + rejects("E_SIGNATURE", "export function f(left: Float, right: Float): boolean { return left === right; }"); +}); + +test("a race may not consume randomness", () => { + rejects( + "E_RACE_EFFECT", + `function draw(): Int | undefined { + return random.nextU32(); +} + +export function f(): Int { + return task.race([(): Int | undefined => draw()]) ?? 0; +}`, + ); +}); + +test("a race may only perform idempotent requests", () => { + rejects( + "E_RACE_EFFECT", + `function send(): string | undefined { + const response = http.request({ + method: "POST", + url: "https://example.test/", + headers: [], + body: "", + timeoutMillis: 1000, + }); + + return response === undefined ? undefined : response.body; +} + +export function f(): string { + return task.race([(): string | undefined => send()]) ?? ""; +}`, + ); +}); diff --git a/engine/tests/switch.spec.ts b/engine/tests/switch.spec.ts new file mode 100644 index 000000000..e91339b9d --- /dev/null +++ b/engine/tests/switch.spec.ts @@ -0,0 +1,116 @@ +/** + * What a `switch` does to the scope around it. + * + * A case is a branch like any other, so what it leaves behind has to reach the code after the + * switch — the same join `if`/`else` takes, and the same one a loop's fixpoint takes. The checker + * used to compute each case's ending state for the exhaustiveness check and then throw it away, + * restoring the state from before the switch: an assignment inside a case was invisible + * afterwards, and a range proven on top of that was a range the program exceeds. + * + * The random program generator found it (`docs/fuzzing.md`), on a `switch` inside a loop whose + * accumulator the checker proved unchanged while the interpreter multiplied it seven times. These + * tests pin the join itself, so the finding survives the seed that produced it. + */ + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { codesOf } from "./helpers.ts"; + +function rejects(code: string, body: string): void { + const codes = codesOf({ "utility.ts": body }); + assert.ok(codes.includes(code), `expected ${code}, got ${codes.join(", ") || "no diagnostics"}`); +} + +function accepts(body: string): void { + const codes = codesOf({ "utility.ts": body }); + assert.deepEqual(codes, [], `expected no diagnostics, got ${codes.join(", ")}`); +} + +test("an assignment inside a case is live after the switch", () => { + rejects( + "E_RETURN_TYPE", + `type Kind = "a" | "b"; + +export function f(kind: Kind): IntRange<0, 0> { + let x: IntRange<0, 7> = 0; + + switch (kind) { + case "a": + x = 7; + break; + default: + break; + } + + return x; +}`, + ); +}); + +test("a case's assignment reaches the next iteration of the enclosing loop", () => { + rejects( + "E_RETURN_TYPE", + `type Kind = "a" | "b"; + +export function f(kind: Kind, items: List): IntRange<1, 1> { + let total: IntRange<1, 64> = 1; + + for (let i = 0; i < items.length; i++) { + switch (kind) { + case "a": + total = int.min(total * 2, 64); + break; + default: + break; + } + } + + return total; +}`, + ); +}); + +test("the join covers every case, not only the last one", () => { + rejects( + "E_RETURN_TYPE", + `type Kind = "a" | "b" | "c"; + +export function f(kind: Kind): IntRange<0, 2> { + let x: IntRange<0, 9> = 0; + + switch (kind) { + case "a": + x = 1; + break; + case "b": + x = 2; + break; + default: + x = 9; + break; + } + + return x; +}`, + ); +}); + +test("a case that returns contributes nothing to the state after the switch", () => { + accepts( + `type Kind = "a" | "b"; + +export function f(kind: Kind): IntRange<0, 1> { + let x: IntRange<0, 9> = 0; + + switch (kind) { + case "a": + x = 9; + return 1; + default: + break; + } + + return x; +}`, + ); +}); diff --git a/engine/tests/translation.spec.ts b/engine/tests/translation.spec.ts new file mode 100644 index 000000000..96e7e0773 --- /dev/null +++ b/engine/tests/translation.spec.ts @@ -0,0 +1,67 @@ +/** + * Translation validation. + * + * Every Core pass has to preserve meaning, so the reference interpreter runs the Core before and + * after the pass over generated inputs and the results must be identical. This is what makes it + * safe for the optimizer to raise a loop into a fold, or to fold a constant away. + */ + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { checkProgram } from "../src/core/check.ts"; +import { Diagnostics } from "../src/diagnostics.ts"; +import { Interpreter } from "../src/interp/interp.ts"; +import { link } from "../src/link/link.ts"; +import { optimize } from "../src/optimize/optimize.ts"; +import { threadCapabilities } from "../src/analysis/capabilities.ts"; +import { loadModules } from "../src/project.ts"; +import type { Value } from "../src/values.ts"; +import { valuesEqual } from "../src/values.ts"; +import { withProject } from "./helpers.ts"; + +const SUM = `export function sumDigits(value: Digits): Int { + let total: Int = 0; + + for (const point of str.codePoints(value)) { + total = total + (point - 48); + } + + return total; +}`; + +const MAPPED = `export function doubled(values: List>): List { + return seq.map(values, (item: IntRange<0, 100>): Int => item * 2); +}`; + +function programsOf(source: string) { + return withProject({ "utility.ts": source }, (root) => { + const diagnostics = new Diagnostics(); + const modules = loadModules(root, diagnostics); + const { program } = checkProgram(modules, diagnostics); + diagnostics.throwIfErrors(); + const threaded = threadCapabilities(program); + return { before: link(threaded), after: link(optimize(threaded)) }; + }); +} + +function sameOnInputs(source: string, fn: string, inputs: readonly Value[][]): void { + const { before, after } = programsOf(source); + for (const args of inputs) { + const left = new Interpreter(before).call(fn, args); + const right = new Interpreter(after).call(fn, args); + assert.ok( + valuesEqual(left, right), + `optimizing changed the result of ${fn}(${JSON.stringify(args.map(String))})`, + ); + } +} + +test("a raised loop and the original agree on generated inputs", () => { + const inputs: Value[][] = ["", "0", "12345", "99999999", "0123456789"].map((value) => [value]); + sameOnInputs(SUM, "utility::sumDigits", inputs); +}); + +test("constant folding and dead code elimination preserve meaning", () => { + const inputs: Value[][] = [[[]], [[1n, 2n, 3n]], [[0n, 100n]]]; + sameOnInputs(MAPPED, "utility::doubled", inputs); +}); diff --git a/engine/toolchain.lock.json b/engine/toolchain.lock.json new file mode 100644 index 000000000..6658b8362 --- /dev/null +++ b/engine/toolchain.lock.json @@ -0,0 +1,45 @@ +{ + "$comment": "Every tool whose version can change generated output or the verification of it. Pinned so a formatter upgrade cannot silently rewrite a golden file.", + "node": "v22.22.2", + "engine": "0.1.0", + "dependencies": { + "oxc-parser": "0.150.0" + }, + "devDependencies": { + "@types/node": "24.13.6", + "prettier": "3.8.1", + "typescript": "5.9.3" + }, + "targets": { + "typescript": { + "baseline": "ES2020 on Node 20", + "formatter": "prettier", + "linters": [ + "tsc --strict --noEmit" + ] + }, + "python": { + "baseline": "Python 3.9", + "formatter": "ruff format", + "formatterVersion": "0.15.8", + "linters": [ + "ruff check", + "pyright" + ] + }, + "go": { + "baseline": "Go 1.21", + "formatter": "gofmt", + "linters": [ + "go vet", + "staticcheck" + ] + } + }, + "measuredWith": { + "node": "v22.22.2", + "python": "Python 3.11.15", + "go": "go version go1.24.7 linux/amd64", + "icu": "78.2" + } +} diff --git a/engine/tsconfig.json b/engine/tsconfig.json new file mode 100644 index 000000000..27b7e7619 --- /dev/null +++ b/engine/tsconfig.json @@ -0,0 +1,18 @@ +{ + "compilerOptions": { + "lib": ["ESNext"], + "target": "ESNext", + "module": "NodeNext", + "moduleResolution": "nodenext", + "allowImportingTsExtensions": true, + "rewriteRelativeImportExtensions": true, + "verbatimModuleSyntax": true, + "erasableSyntaxOnly": true, + "noEmit": true, + "strict": true, + "skipLibCheck": true, + "noFallthroughCasesInSwitch": true, + "noImplicitOverride": true + }, + "include": ["src/**/*.ts", "tests/**/*.ts", "scripts/**/*.ts"] +} diff --git a/tsconfig.json b/tsconfig.json index fc21f1755..6ca7afed8 100644 --- a/tsconfig.json +++ b/tsconfig.json @@ -17,5 +17,8 @@ "noUnusedParameters": true, "noPropertyAccessFromIndexSignature": true, "noImplicitOverride": true - } + }, + // `engine` and `core` are a separate project with their own tsconfig and their own + // verification pipeline; the package's type checking stops at its own sources. + "exclude": ["dist", "node_modules", "engine", "core"] } diff --git a/vite.config.ts b/vite.config.ts index 341200a82..431cbccb4 100644 --- a/vite.config.ts +++ b/vite.config.ts @@ -157,6 +157,13 @@ export default defineConfig({ ".stryker-tmp", ".claude", "CHANGELOG.md", + // `engine` and `core` are a separate, self-contained project with its own toolchain + // (its own tsconfig, its own tests, and the target formatters it runs over its own + // output). It is formatted, linted and tested by `core/package.json`'s `verify`, and + // keeping it out of the package's own passes is what lets it be extracted later + // without carrying this repository's configuration with it. + "engine", + "core", ], singleQuote: false, sortImports: true, @@ -178,7 +185,16 @@ export default defineConfig({ perf: "error", pedantic: "error", }, - ignorePatterns: ["dist", "coverage", "docs", "reports", ".stryker-tmp", ".claude"], + ignorePatterns: [ + "dist", + "coverage", + "docs", + "reports", + ".stryker-tmp", + ".claude", + "engine", + "core", + ], rules: { "eslint/complexity": ["error", { max: 20 }], "eslint/max-lines": "off", @@ -515,7 +531,15 @@ export default defineConfig({ ], }, test: { - exclude: ["**/node_modules/**", "**/dist/**", "**/.stryker-tmp/**", "**/reports/**"], + exclude: [ + "**/node_modules/**", + "**/dist/**", + "**/.stryker-tmp/**", + "**/reports/**", + // The engine runs its own suite through `node --test`; see the note in `fmt` above. + "engine/**", + "core/**", + ], benchmark: { include: ["src/**/*.test.ts"], exclude: ["**/node_modules/**", "**/dist/**", "**/.stryker-tmp/**", "**/reports/**"],