From d3ddaa4b6cbd115d8ac93670b680086d5fea6897 Mon Sep 17 00:00:00 2001 From: Luiz Lima Date: Fri, 25 Sep 2026 21:34:29 -0300 Subject: [PATCH 01/10] feat(schema): declare the hunk suggestion and its optional plan echo A saved plan may carry the suggested class per hunk; --plan never reads it, and a hand-mangled one is dropped instead of failing the plan. --- src/schema/index.ts | 20 ++++++++++++++++++++ test/unify.test.ts | 28 +++++++++++++++++++++++++++- 2 files changed, 47 insertions(+), 1 deletion(-) diff --git a/src/schema/index.ts b/src/schema/index.ts index 719104b..122d588 100644 --- a/src/schema/index.ts +++ b/src/schema/index.ts @@ -232,11 +232,31 @@ export type LockEntry = z.infer; export const TakeSchema = z.enum(["base", "variant", "keep"]); export type Take = z.infer; +/** The class `forge diff` suggests for a hunk (spec 08). A suggestion never decides anything. */ +export const HUNK_CLASSES = ["evolution", "value", "block"] as const; +export const HunkClassSchema = z.enum(HUNK_CLASSES); +export type HunkClass = z.infer; + +export const SuggestedTokenSchema = z.object({ + a: z.string(), // the base side's text for this change + b: z.string(), // the variant side's text + param: z.string(), // "param." +}); + +export const HunkSuggestionSchema = z.object({ + class: HunkClassSchema, + reason: z.string(), + tokens: z.array(SuggestedTokenSchema).optional(), +}); +export type HunkSuggestion = z.infer; + export const PlanHunkSchema = z.object({ hunk: z.number().int().positive(), /** Human echo of what `forge diff` printed. Never read back. */ at: z.string().default(""), take: TakeSchema, + /** Echo of the suggested class when the plan was saved. Never read back; a malformed one is dropped. */ + suggestion: HunkSuggestionSchema.optional().catch(undefined), }); export type PlanHunk = z.infer; diff --git a/test/unify.test.ts b/test/unify.test.ts index c4ff401..9700852 100644 --- a/test/unify.test.ts +++ b/test/unify.test.ts @@ -7,7 +7,7 @@ import YAML from "yaml"; import { exists, gitDirty, loadForge, type Forge } from "../src/core/forge.js"; import { applyPlan, metaDifferences, planFrom, rewriteRecipes, writeUnified } from "../src/core/unify.js"; import { diffIngredients } from "../src/core/variants.js"; -import { UnifyPlanSchema } from "../src/schema/index.js"; +import { HunkSuggestionSchema, UnifyPlanSchema } from "../src/schema/index.js"; import { makeForge, profile, recipe, rule, tmpDir, writeFiles, type ForgeSpec } from "./helpers/forge.js"; const execFileP = promisify(execFile); @@ -569,3 +569,29 @@ describe("metaDifferences — MCP server key order (spec 07, AC 19)", () => { expect(metaDifferences(forge.ingredients.get("mcp/p")!, forge.ingredients.get("mcp/p--acme")!)).toEqual([]); }); }); + +describe("the hunk suggestion in a plan (spec 08 §5.1)", () => { + const plan = (hunk: Record) => ({ + schema: 1, + base: "rule/w", + profile: "acme", + variant: "rule/w--acme", + baseFingerprint: "sha256:a", + variantFingerprint: "sha256:b", + files: [{ file: "rule.md", hunks: [{ hunk: 1, at: "lines 1–1", take: "keep", ...hunk }] }], + }); + const value = { class: "value", reason: "1 token differs", tokens: [{ a: "acme-api", b: "globex-api", param: "param.acme_api" }] }; + + it("accepts the three classes and refuses any other", () => { + expect(HunkSuggestionSchema.safeParse(value).success).toBe(true); + expect(HunkSuggestionSchema.safeParse({ class: "evolution", reason: "prose differs" }).success).toBe(true); + expect(HunkSuggestionSchema.safeParse({ class: "bogus", reason: "x" }).success).toBe(false); + }); + + it("keeps a valid suggestion, drops a malformed one without failing, and accepts a plan without one", () => { + expect(UnifyPlanSchema.parse(plan({ suggestion: value })).files[0].hunks![0].suggestion).toEqual(value); + const mangled = UnifyPlanSchema.parse(plan({ suggestion: { class: "bogus" } })).files[0].hunks![0]; + expect(mangled.suggestion).toBeUndefined(); + expect(UnifyPlanSchema.parse(plan({})).files[0].hunks![0]).toEqual({ hunk: 1, at: "lines 1–1", take: "keep" }); + }); +}); From acf97f951a9962bd954595ef2b7273001597bc3e Mon Sep 17 00:00:00 2001 From: Luiz Lima Date: Fri, 25 Sep 2026 21:34:52 -0300 Subject: [PATCH 02/10] refactor(diff): export the LCS so the word diff reuses it One LCS for lines and for words, with the same tie-break; diffOps behaves as before. --- src/core/diff.ts | 7 +++++-- test/diff.test.ts | 13 ++++++++++++- 2 files changed, 17 insertions(+), 3 deletions(-) diff --git a/src/core/diff.ts b/src/core/diff.ts index 12a0f00..be556c8 100644 --- a/src/core/diff.ts +++ b/src/core/diff.ts @@ -33,8 +33,11 @@ export function splitLines(s: string): Split { * that is only line endings or a BOM produces no ops other than "same". */ export function diffOps(a: string, b: string): DiffOp[] { - const A = splitLines(a).lines; - const B = splitLines(b).lines; + return lcsOps(splitLines(a).lines, splitLines(b).lines); +} + +/** LCS over two lists of strings, compared exactly; shared by the line diff and the word diff (spec 08 §6.2). */ +export function lcsOps(A: string[], B: string[]): DiffOp[] { const n = A.length; const m = B.length; const dp: number[][] = Array.from({ length: n + 1 }, () => new Array(m + 1).fill(0)); diff --git a/test/diff.test.ts b/test/diff.test.ts index d411af8..6e6af5b 100644 --- a/test/diff.test.ts +++ b/test/diff.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "vitest"; -import { diffLines, diffOps, renderDiff, splitLines } from "../src/core/diff.js"; +import { diffLines, diffOps, lcsOps, renderDiff, splitLines } from "../src/core/diff.js"; describe("diffLines", () => { it("reports nothing for identical text", () => { @@ -243,3 +243,14 @@ describe("renderDiff", () => { } }); }); + +describe("lcsOps", () => { + it("diffs two token lists with the line diff's tie-break", () => { + expect(lcsOps(["a", " ", "b"], ["a", " ", "c"])).toEqual([ + { kind: "same", line: "a" }, + { kind: "same", line: " " }, + { kind: "del", line: "b" }, + { kind: "add", line: "c" }, + ]); + }); +}); From 367080d57d74e915bd274e2ddffecdc94518849f Mon Sep 17 00:00:00 2001 From: Luiz Lima Date: Fri, 25 Sep 2026 21:36:27 -0300 Subject: [PATCH 03/10] feat(classify): suggest evolution, value or block for each hunk A pure heuristic over the real hunk shapes: one-sided lines are blocks, a swap of identifier-like tokens is a value with a param., and anything else is evolution. --- src/core/classify.ts | 120 ++++++++++++++++++++++ test/classify.test.ts | 227 ++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 347 insertions(+) create mode 100644 src/core/classify.ts create mode 100644 test/classify.test.ts diff --git a/src/core/classify.ts b/src/core/classify.ts new file mode 100644 index 0000000..685e5ae --- /dev/null +++ b/src/core/classify.ts @@ -0,0 +1,120 @@ +import type { HunkSuggestion } from "../schema/index.js"; +import { lcsOps, type Hunk } from "./diff.js"; + +/** + * Suggests a class for one hunk (spec 08 §6): `evolution` (one side is a newer text), `value` + * (a short client token swapped inside shared prose) or `block` (lines only one side has). + * Pure and deterministic. A suggestion never decides anything; `forge unify` still asks. + */ + +export type ClassifiedHunk = Hunk & { suggestion: HunkSuggestion }; + +/** A token longer than this is not a client identifier (spec 08 §6.2). */ +const MAX_TOKEN_CHARS = 40; +/** Bounds the O(n·m) token LCS; no real rule line comes near it. */ +const MAX_LINE_TOKENS = 400; +/** Integers this short are step numbers and list markers, not client data. */ +const NUMBERING = /^[0-9]{1,2}$/; + +const TOKEN = /\s+|[\p{L}\p{N}_@-]+(?:[./:]+[\p{L}\p{N}_@-]+)*|\S/gu; +const WORD = /^[\p{L}\p{N}_@-]+(?:[./:]+[\p{L}\p{N}_@-]+)*$/u; +const WHITESPACE = /^\s+$/; + +/** A line split into whitespace runs, words (which may hold `.`, `/`, `:` between word characters) and single other characters. */ +export function tokenize(line: string): string[] { + return line.match(TOKEN) ?? []; +} + +/** A word short enough, not a step number, and shaped like a name, path, version or port (spec 08 §6.2). */ +export function isIdentifierLike(token: string): boolean { + if (!WORD.test(token) || token.length > MAX_TOKEN_CHARS || NUMBERING.test(token)) return false; + return /[-./@:]/.test(token) || /[0-9]/.test(token) || /[\p{L}\p{N}]_[\p{L}\p{N}]/u.test(token) || /\p{Ll}\p{Lu}/u.test(token); +} + +/** `param.` of a token's base-side text; `_`-separated because `substitute` keys are `[A-Za-z0-9_.]` (spec 08 §6.4). */ +export function paramSlug(text: string): string { + let slug = text + .normalize("NFD") + .replace(/\p{M}/gu, "") + .toLowerCase() + .replace(/[^a-z0-9]+/g, "_") + .replace(/^_+|_+$/g, ""); + slug = slug.slice(0, MAX_TOKEN_CHARS).replace(/_+$/, ""); + return `param.${slug || "value"}`; +} + +const evolution = (reason: string): HunkSuggestion => ({ class: "evolution", reason }); + +interface Region { + a: string[]; + b: string[]; +} + +/** Maximal runs of changed tokens between two lines; whitespace runs compare equal whatever their length. */ +function regions(a: string[], b: string[]): Region[] { + const key = (t: string) => (WHITESPACE.test(t) ? " " : t); + const out: Region[] = []; + let current: Region | null = null; + let i = 0; + let j = 0; + for (const op of lcsOps(a.map(key), b.map(key))) { + if (op.kind === "same") { + current = null; + i++; + j++; + continue; + } + if (!current) out.push((current = { a: [], b: [] })); + if (op.kind === "del") current.a.push(a[i++]); + else current.b.push(b[j++]); + } + return out; +} + +const text = (tokens: string[]) => tokens.join("").replace(/\s+/g, " ").trim(); +const words = (tokens: string[]) => tokens.filter((t) => !WHITESPACE.test(t)); + +export function classifyHunk(h: Hunk): HunkSuggestion { + // 1. Drop equal lines at both ends: the context line `forceTrailingHunk` adds, nothing else. + let a = h.a.lines; + let b = h.b.lines; + while (a.length && b.length && a[0] === b[0]) [a, b] = [a.slice(1), b.slice(1)]; + while (a.length && b.length && a[a.length - 1] === b[b.length - 1]) [a, b] = [a.slice(0, -1), b.slice(0, -1)]; + + if (!a.length && !b.length) return evolution("final newline only"); + if (!a.length) return { class: "block", reason: "only in the variant" }; + if (!b.length) return { class: "block", reason: "only in the base" }; + if (Boolean(h.a.noEofNewline) !== Boolean(h.b.noEofNewline)) return evolution("final newline differs"); + if (a.length !== b.length) return evolution("line counts differ"); + + const changed: Array<{ a: string; b: string }> = []; + for (let k = 0; k < a.length; k++) { + const ta = tokenize(a[k]); + const tb = tokenize(b[k]); + if (ta.length > MAX_LINE_TOKENS || tb.length > MAX_LINE_TOKENS) return evolution("line too long to compare"); + for (const r of regions(ta, tb)) { + const wa = words(r.a); + const wb = words(r.b); + if (!wa.length && !wb.length) continue; // whitespace only + if (!wa.length || !wb.length) return evolution("words added or removed"); + const odd = [...wa, ...wb].filter((t) => !isIdentifierLike(t)); + if (odd.length) return evolution(odd.every((t) => NUMBERING.test(t)) ? "numbering differs" : "prose differs"); + changed.push({ a: text(r.a), b: text(r.b) }); + } + } + if (!changed.length) return evolution("whitespace only"); + + const seen = new Set(); + const slugCount = new Map(); + const tokens: NonNullable = []; + for (const c of changed) { + const id = JSON.stringify([c.a, c.b]); + if (seen.has(id)) continue; + seen.add(id); + const slug = paramSlug(c.a); + const n = (slugCount.get(slug) ?? 0) + 1; + slugCount.set(slug, n); + tokens.push({ a: c.a, b: c.b, param: n === 1 ? slug : `${slug}_${n}` }); + } + return { class: "value", reason: tokens.length === 1 ? "1 token differs" : `${tokens.length} tokens differ`, tokens }; +} diff --git a/test/classify.test.ts b/test/classify.test.ts new file mode 100644 index 0000000..966adb2 --- /dev/null +++ b/test/classify.test.ts @@ -0,0 +1,227 @@ +import { describe, expect, it } from "vitest"; +import { classifyHunk, isIdentifierLike, paramSlug, tokenize } from "../src/core/classify.js"; +import { diffLines } from "../src/core/diff.js"; +import type { HunkSuggestion } from "../src/schema/index.js"; + +/** The one hunk between two texts, built by the real diff (spec 08 §9.1: never by hand). */ +function hunkOf(a: string, b: string) { + const hunks = diffLines(a, b); + expect(hunks).toHaveLength(1); + return hunks[0]; +} +const classify = (a: string, b: string) => classifyHunk(hunkOf(a, b)); +const brief = (s: HunkSuggestion) => `${s.class}: ${s.reason}`; + +const REASONS = new Set([ + "only in the base", + "only in the variant", + "final newline only", + "final newline differs", + "line counts differ", + "line too long to compare", + "words added or removed", + "numbering differs", + "prose differs", + "whitespace only", +]); +const allowed = (s: HunkSuggestion) => REASONS.has(s.reason) || /^(1 token differs|[0-9]+ tokens differ)$/.test(s.reason); + +describe("tokenize", () => { + it("splits whitespace runs, words and single other characters", () => { + expect(tokenize("a b\tc")).toEqual(["a", " ", "b", "\t", "c"]); + expect(tokenize("acme-api.")).toEqual(["acme-api", "."]); + expect(tokenize("`x`")).toEqual(["`", "x", "`"]); + expect(tokenize("| a |")).toEqual(["|", " ", "a", " ", "|"]); + }); + + it("keeps URLs, e-mails, scopes and dotted names as one word", () => { + for (const w of ["https://acme.dev/docs", "ops@acme.dev", "@acme/npm", "Directory.Build.props", "NPM_PAT"]) expect(tokenize(w)).toEqual([w]); + }); + + it("reads Unicode letters as word characters", () => { + expect(tokenize("ação-api")).toEqual(["ação-api"]); + }); +}); + +describe("isIdentifierLike", () => { + it("accepts names, paths, versions and ports", () => { + for (const t of ["acme-api", "package.json", "ops@acme.dev", "a/b", "x:y", "v2", "555", "44301", "NPM_PAT", "getAcmeUser"]) expect(isIdentifierLike(t), t).toBe(true); + }); + + it("refuses plain words, step numbers, emphasis and non-words", () => { + for (const t of ["Acme", "acme", "5", "55", "_TODO_", "`", "|", "step"]) expect(isIdentifierLike(t), t).toBe(false); + }); + + it("stops at 40 characters", () => { + expect(isIdentifierLike("a-" + "x".repeat(38))).toBe(true); + expect(isIdentifierLike("a-" + "x".repeat(39))).toBe(false); + }); +}); + +describe("paramSlug", () => { + it("slugs the base-side text with underscores", () => { + expect(paramSlug("package.json")).toBe("param.package_json"); + expect(paramSlug("acme-portal-api")).toBe("param.acme_portal_api"); + expect(paramSlug("44301")).toBe("param.44301"); + expect(paramSlug("https://acme.dev/docs")).toBe("param.https_acme_dev_docs"); + }); + + it("strips diacritics, falls back to value, and truncates to 40 characters", () => { + expect(paramSlug("ação-api")).toBe("param.acao_api"); + expect(paramSlug("---")).toBe("param.value"); + expect(paramSlug("x".repeat(60))).toBe("param." + "x".repeat(40)); + expect(paramSlug("x".repeat(39) + "-y")).toBe("param." + "x".repeat(39)); + }); +}); + +describe("classifyHunk — the seven shapes of spec 01 §2, synthetic", () => { + it("workflow: a TODO replaced by a reference is evolution", () => { + expect(brief(classify("# W\n_TODO_: decide who reviews.\n", "# W\nSee `review-posture.md`.\n"))).toBe("evolution: prose differs"); + }); + + it("commit-conventions: a clause added is evolution", () => { + expect(brief(classify("- No Co-Authored-By trailer.\n", "- No Co-Authored-By trailer, unless the user explicitly asks.\n"))).toBe( + "evolution: words added or removed", + ); + }); + + it("branch-and-pr: a version file name swapped is a value", () => { + const s = classify("# B\nBump `package.json` before the PR.\n", "# B\nBump `Directory.Build.props` before the PR.\n"); + expect(brief(s)).toBe("value: 1 token differs"); + expect(s.tokens).toEqual([{ a: "package.json", b: "Directory.Build.props", param: "param.package_json" }]); + }); + + it("commit-identity: a registry described away is evolution", () => { + expect( + brief(classify("The `NPM_PAT` token reads the `@acme-registry` feed.\n", "The `NPM_PAT` token reads the feed, if/when there is one.\n")), + ).toBe("evolution: words added or removed"); + }); + + it("review-posture: a client row added to a table is a block", () => { + const row = "| acme-api | node-cli-reviewer |\n"; + expect(brief(classify(row, row + "| acme-desktop | desktop-reviewer |\n"))).toBe("block: only in the variant"); + }); + + it("repo-discovery: a whole section only one client has is a block", () => { + const common = "# Repos\nShared text.\n"; + expect(brief(classify(common + "## Desktop\nacme-desktop uses pnpm.\nBuild with electron.\n", common))).toBe("block: only in the base"); + }); + + it("open-pr: a step renumbered next to a client mention is evolution", () => { + expect(brief(classify("Step 6: open the PR for acme-desktop.\n", "Step 5: open the PR.\n"))).toBe("evolution: numbering differs"); + }); +}); + +describe("classifyHunk — edge cases (spec 08 §7)", () => { + it("1. a newline-only difference", () => { + expect(brief(classify("a\nb\n", "a\nb"))).toBe("evolution: final newline only"); + }); + + it("2. a line appended after an unterminated last line is a block, whatever the structural kind", () => { + const h = hunkOf("a\nb", "a\nb\nc\n"); + expect(h.kind).toBe("inline"); + expect(brief(classifyHunk(h))).toBe("block: only in the variant"); + }); + + it("3. a token swap plus a final-newline change is evolution", () => { + expect(brief(classify("x\nacme-api", "x\nglobex-api\n"))).toBe("evolution: final newline differs"); + }); + + it("4. step numbers and list markers", () => { + expect(brief(classify("Step 6: open the PR.\n", "Step 5: open the PR.\n"))).toBe("evolution: numbering differs"); + expect(brief(classify("1. first\n", "2. first\n"))).toBe("evolution: numbering differs"); + }); + + it("5. numbering plus a client mention: the first region names the reason", () => { + expect(brief(classify("Step 6: open the PR for acme-desktop.\n", "Step 5: open the PR.\n"))).toBe("evolution: numbering differs"); + }); + + it("6. a value and prose in one line", () => { + expect(brief(classify("Use acme-api now.\n", "Use globex-api today.\n"))).toBe("evolution: prose differs"); + }); + + it("7. a changed row plus added rows is evolution; added rows alone are a block", () => { + const base = "| a | acme-api |\n"; + expect(brief(classify(base, "| a | globex-api |\n| b | globex-web |\n"))).toBe("evolution: line counts differ"); + }); + + it("8. known false positives: hyphenated words, abbreviations and years", () => { + expect(classify("A well-known rule.\n", "A long-lived rule.\n").class).toBe("value"); + expect(classify("Use e.g. this.\n", "Use i.e. this.\n").class).toBe("value"); + expect(classify("Since 2025.\n", "Since 2026.\n").class).toBe("value"); + }); + + it("9. known false negatives: plain client names", () => { + for (const [a, b] of [["acme", "globex"], ["Acme", "Globex"], ["ACME", "GLOBEX"]]) + expect(brief(classify(`Owned by ${a}.\n`, `Owned by ${b}.\n`))).toBe("evolution: prose differs"); + }); + + it("10. a duplicated token is listed once", () => { + const s = classify("acme-api and acme-api\n", "globex-api and globex-api\n"); + expect(brief(s)).toBe("value: 1 token differs"); + expect(s.tokens).toHaveLength(1); + }); + + it("11. a slug collision within one hunk gets a numeric suffix", () => { + const s = classify("acme-api then acme_api\n", "x-1 then x-2\n"); + expect(s.tokens?.map((t) => t.param)).toEqual(["param.acme_api", "param.acme_api_2"]); + expect(s.reason).toBe("2 tokens differ"); + }); + + it("12. an over-long token is not a client identifier", () => { + expect(brief(classify(`Use a-${"x".repeat(39)} here.\n`, "Use acme-api here.\n"))).toBe("evolution: prose differs"); + }); + + it("13. an over-long line is not compared word by word", () => { + const long = (w: string) => Array.from({ length: 201 }, () => w).join(" ") + "\n"; + expect(brief(classify(long("acme-api"), long("globex-api")))).toBe("evolution: line too long to compare"); + }); + + it("15. the classifier sees what the diff sees", () => { + expect(brief(classify("café-api\n", "cafe-api\n"))).toBe("value: 1 token differs"); + }); +}); + +describe("classifyHunk — whitespace and markdown (spec 08 §6.2 consequences)", () => { + it("table padding around a swapped value", () => { + expect(brief(classify("| acme-api |\n", "| globex-api |\n"))).toBe("value: 1 token differs"); + }); + it("a tab replaced by a space", () => { + expect(brief(classify("a\tb\n", "a b\n"))).toBe("evolution: whitespace only"); + }); + it("backticks kept, content swapped", () => { + expect(brief(classify("`acme-api`\n", "`globex-api`\n"))).toBe("value: 1 token differs"); + }); + it("backticks removed", () => { + expect(brief(classify("`acme-api`\n", "acme-api\n"))).toBe("evolution: words added or removed"); + }); + it("a table cell added", () => { + expect(brief(classify("| a |\n", "| a | b |\n"))).toBe("evolution: words added or removed"); + }); + it("an empty line against text", () => { + expect(brief(classify("x\n\ny\n", "x\ntext\ny\n"))).toBe("evolution: words added or removed"); + }); +}); + +describe("classifyHunk — contract (spec 08 AC 2)", () => { + it("returns only reasons from the closed set, and tokens only for a value", () => { + const corpus: Array<[string, string]> = [ + ["_TODO_: decide.\n", "See `x.md`.\n"], + ["Bump `package.json`.\n", "Bump `a.props`.\n"], + ["r\n", "r\nq\n"], + ["a\nb\n", "a\nb"], + ["Step 6.\n", "Step 5.\n"], + ["a\tb\n", "a b\n"], + ["x\nacme-api", "x\nglobex-api\n"], + ["one\n", "one\ntwo\nthree\n"], + ]; + for (const [a, b] of corpus) { + for (const h of diffLines(a, b)) { + const s = classifyHunk(h); + expect(allowed(s), s.reason).toBe(true); + expect(s.tokens !== undefined, s.reason).toBe(s.class === "value"); + for (const t of s.tokens ?? []) expect(t.param).toMatch(/^param\.[a-z0-9_]+$/); + } + } + }); +}); From 0fd1875e24ec0b5b72cc9bd6163c61bf5e99a87f Mon Sep 17 00:00:00 2001 From: Luiz Lima Date: Fri, 25 Sep 2026 21:37:45 -0300 Subject: [PATCH 04/10] feat(forge): classify every hunk and count the classes per variant diffIngredients is the single place hunks are classified, so forge diff, forge variants and --save-plan agree; a one-sided file counts as a block, keeping the sum at distance.hunks. --- src/core/unify.ts | 2 +- src/core/variants.ts | 27 ++++++++++++++++++--------- test/unify.test.ts | 21 +++++++++++++++++++-- test/variants.test.ts | 26 ++++++++++++++++++++++++++ 4 files changed, 64 insertions(+), 12 deletions(-) diff --git a/src/core/unify.ts b/src/core/unify.ts index 784a133..2d5282a 100644 --- a/src/core/unify.ts +++ b/src/core/unify.ts @@ -74,7 +74,7 @@ export async function planFrom( await assertTextMergeable(base, variant); const files: PlanFile[] = []; for (const f of diff.files) { - files.push({ file: f.file, hunks: f.hunks.map((h, i) => ({ hunk: i + 1, at: hunkAt(h), take: "keep" as const })) }); + files.push({ file: f.file, hunks: f.hunks.map((h, i) => ({ hunk: i + 1, at: hunkAt(h), take: "keep" as const, suggestion: h.suggestion })) }); } for (const file of diff.onlyInBase) files.push({ file, onlyIn: "base", take: "keep" }); for (const file of diff.onlyInVariant) files.push({ file, onlyIn: "variant", take: "keep" }); diff --git a/src/core/variants.ts b/src/core/variants.ts index 03c3d6b..4d42ca1 100644 --- a/src/core/variants.ts +++ b/src/core/variants.ts @@ -1,9 +1,10 @@ import { promises as fs } from "node:fs"; import path from "node:path"; +import { classifyHunk, type ClassifiedHunk } from "./classify.js"; import { diffLines, splitLines, type Hunk } from "./diff.js"; import { fingerprintDir } from "./fingerprint.js"; import { listFiles, type Forge, type LoadedIngredient } from "./forge.js"; -import type { IngredientRef } from "../schema/index.js"; +import type { HunkClass, IngredientRef } from "../schema/index.js"; export interface Distance { lines: number; @@ -17,6 +18,8 @@ export interface VariantEntry { ref: IngredientRef; profile: string; distance: Distance; + /** Hunks by suggested class; a file present on one side only counts as one `block` (spec 08 §4.2). */ + classes: Record; } export interface VariantGroup { @@ -37,7 +40,7 @@ export interface VariantReport { export interface FileDiff { file: string; - hunks: Hunk[]; + hunks: ClassifiedHunk[]; } export interface IngredientDiff { @@ -75,7 +78,8 @@ export async function diffIngredients(base: LoadedIngredient, variant: LoadedIng for (const rel of [...a.keys()].sort()) { const other = b.get(rel); if (other === undefined) continue; - const hunks = diffLines(a.get(rel)!, other); + // Classified here and nowhere else, so forge diff, forge variants and --save-plan agree (spec 08 §5.2). + const hunks = diffLines(a.get(rel)!, other).map((h) => ({ ...h, suggestion: classifyHunk(h) })); if (hunks.length) files.push({ file: rel, hunks }); } return { @@ -85,7 +89,7 @@ export async function diffIngredients(base: LoadedIngredient, variant: LoadedIng }; } -async function distanceOf(base: LoadedIngredient, variant: LoadedIngredient): Promise { +async function measure(base: LoadedIngredient, variant: LoadedIngredient): Promise<{ distance: Distance; classes: Record }> { const d = await diffIngredients(base, variant); const baseFiles = await filesOf(base); const variantFiles = await filesOf(variant); @@ -106,11 +110,16 @@ async function distanceOf(base: LoadedIngredient, variant: LoadedIngredient): Pr const lines = d.files.reduce((n, f) => n + lineCount(f.hunks), 0) + oneSidedLines; const bodyDiffers = hunks > 0; const sameFingerprint = (await fingerprintDir(base.dir)) === (await fingerprintDir(variant.dir)); + const classes: Record = { evolution: 0, value: 0, block: d.onlyInBase.length + d.onlyInVariant.length }; + for (const f of d.files) for (const h of f.hunks) classes[h.suggestion.class]++; return { - lines, - hunks, - sameBodyDifferentMeta: !bodyDiffers && !sameFingerprint, - identicalAfterNormalization: !bodyDiffers && sameFingerprint, + distance: { + lines, + hunks, + sameBodyDifferentMeta: !bodyDiffers && !sameFingerprint, + identicalAfterNormalization: !bodyDiffers && sameFingerprint, + }, + classes, }; } @@ -131,7 +140,7 @@ export async function listVariants(forge: Forge): Promise { orphans.push({ ref: ing.ref, profile, missingBase: baseRef }); continue; } - const entry: VariantEntry = { ref: ing.ref, profile, distance: await distanceOf(base, ing) }; + const entry: VariantEntry = { ref: ing.ref, profile, ...(await measure(base, ing)) }; groups.set(baseRef, [...(groups.get(baseRef) ?? []), entry]); } return { diff --git a/test/unify.test.ts b/test/unify.test.ts index 9700852..6a5afb5 100644 --- a/test/unify.test.ts +++ b/test/unify.test.ts @@ -75,8 +75,8 @@ function fakeDiff() { return { files: [ { file: "rule.md", hunks: [ - { kind: "inline" as const, a: { start: 2, lines: ["old"] }, b: { start: 2, lines: ["new"] } }, - { kind: "block" as const, a: { start: 5, lines: [] }, b: { start: 5, lines: ["added"] } }, + { kind: "inline" as const, a: { start: 2, lines: ["old"] }, b: { start: 2, lines: ["new"] }, suggestion: { class: "evolution" as const, reason: "prose differs" } }, + { kind: "block" as const, a: { start: 5, lines: [] }, b: { start: 5, lines: ["added"] }, suggestion: { class: "block" as const, reason: "only in the variant" } }, ] }, ], onlyInBase: ["gone.md"], @@ -109,6 +109,8 @@ describe("planFrom", () => { expect(paired.hunks!.map((h) => h.hunk)).toEqual([1, 2]); expect(paired.hunks![0].at).toBe("lines 2–2"); expect(paired.hunks![1].at).toBe("after line 4"); + expect(paired.hunks!.map((h) => h.suggestion?.class)).toEqual(["evolution", "block"]); + for (const f of plan.files.filter((f) => f.onlyIn)) expect(f).not.toHaveProperty("suggestion"); expect(plan.files.find((f) => f.file === "gone.md")).toMatchObject({ onlyIn: "base", take: "keep" }); expect(plan.files.find((f) => f.file === "extra.md")).toMatchObject({ onlyIn: "variant", take: "keep" }); }); @@ -130,6 +132,21 @@ async function scenario(baseFiles: Record, variantFiles: Record< return { base, variant, diff }; } +describe("applyPlan ignores the suggestion (spec 08 §4.3)", () => { + it("gives the same result with the suggestions, without them, and with a mangled one", async () => { + const { base, variant, diff } = await scenario({ "rule.md": "a\nuse acme-api\nc\n" }, { "rule.md": "a\nuse globex-api\nc\n" }); + const plan = await planFrom(base, variant, diff, "acme"); + plan.files[0].hunks![0].take = "variant"; + expect(plan.files[0].hunks![0].suggestion?.class).toBe("value"); + const stripped = structuredClone(plan); + delete stripped.files[0].hunks![0].suggestion; + const mangled = UnifyPlanSchema.parse({ ...structuredClone(plan), files: [{ ...plan.files[0], hunks: [{ ...plan.files[0].hunks![0], suggestion: { class: "bogus" } }] }] }); + const r = await applyPlan(base, variant, diff, plan); + expect(await applyPlan(base, variant, diff, stripped)).toEqual(r); + expect(await applyPlan(base, variant, diff, mangled)).toEqual(r); + }); +}); + describe("applyPlan — paired files", () => { it("takes the variant's side for a chosen hunk and the base's for the rest", async () => { // base rule.md: "a\nold\nc\n" diff --git a/test/variants.test.ts b/test/variants.test.ts index e3193b9..bcbd261 100644 --- a/test/variants.test.ts +++ b/test/variants.test.ts @@ -2,6 +2,7 @@ import { afterEach, describe, expect, it } from "vitest"; import { promises as fs } from "node:fs"; import { loadForge } from "../src/core/forge.js"; import { diffIngredients, listVariants, profileOf } from "../src/core/variants.js"; +import { classifyHunk } from "../src/core/classify.js"; import { makeForge, rule, tmpDir, type ForgeSpec } from "./helpers/forge.js"; const cleanups: Array<() => Promise> = []; @@ -176,3 +177,28 @@ describe("diffIngredients", () => { expect(d.files[0].hunks.map((h) => h.kind)).toEqual(["block"]); }); }); + +describe("hunk classes (spec 08 §4.2, §5.2)", () => { + it("classifies every hunk diffIngredients returns", async () => { + const forge = await forgeWith([rule("x", "a\nuse acme-api\n"), rule("x--acme", "a\nuse globex-api\nmore\n", { as: "x" })]); + const d = await diffIngredients(forge.ingredients.get("rule/x")!, forge.ingredients.get("rule/x--acme")!); + for (const f of d.files) for (const h of f.hunks) expect(h.suggestion).toEqual(classifyHunk(h)); + }); + + it("counts the classes per variant, a one-sided file as block, and the counts add up to distance.hunks", async () => { + const forge = await forgeWith([ + { meta: { type: "rule", name: "x" }, files: { "rule.md": "Bump `package.json`.\nshared\nStep 6.\nkeep\n" } }, + { meta: { type: "rule", name: "x--acme", as: "x" }, files: { "rule.md": "Bump `a.props`.\nshared\nStep 5.\nkeep\nnew\n", "extra.md": "only here\n" } }, + rule("y", "same\n"), + rule("y--acme", "same\n", { as: "y", targets: ["kiro"] }), + rule("z", "one\ntwo\n"), + rule("z--acme", "one\r\ntwo\r\n", { as: "z" }), + ]); + const { groups } = await listVariants(forge); + const entry = (ref: string) => groups.flatMap((g) => g.variants).find((v) => v.ref === ref)!; + expect(entry("rule/x--acme").classes).toEqual({ evolution: 1, value: 1, block: 2 }); + expect(entry("rule/y--acme").classes).toEqual({ evolution: 0, value: 0, block: 0 }); + expect(entry("rule/z--acme").classes).toEqual({ evolution: 0, value: 0, block: 0 }); + for (const v of groups.flatMap((g) => g.variants)) expect(v.classes.evolution + v.classes.value + v.classes.block).toBe(v.distance.hunks); + }); +}); From a986fb9e79eb2bf23ba69b03181dcfc467fd785f Mon Sep 17 00:00:00 2001 From: Luiz Lima Date: Fri, 25 Sep 2026 21:39:13 -0300 Subject: [PATCH 05/10] feat(cli): show the suggested class in forge diff and forge variants Each hunk header gains its class and reason (and the suggested params for a value), and each variant its counts by class, after the unchanged prefixes. --- src/cli.ts | 18 +++++++++-- test/cli.test.ts | 79 ++++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 94 insertions(+), 3 deletions(-) diff --git a/src/cli.ts b/src/cli.ts index d7b3169..8bd3f5e 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -20,7 +20,7 @@ import { type RecipeCascadeResult, type WriteJournal, } from "./core/unify.js"; -import { UnifyPlanSchema, type IngredientRef, type Take, type UnifyPlan } from "./schema/index.js"; +import { HUNK_CLASSES, UnifyPlanSchema, type HunkClass, type HunkSuggestion, type IngredientRef, type Take, type UnifyPlan } from "./schema/index.js"; process.stdout.on("error", (e: NodeJS.ErrnoException) => { if (e.code === "EPIPE") process.exit(0); }); @@ -177,7 +177,7 @@ forge if (!groups.length && !orphans.length) return console.log(" no variants"); for (const g of groups) { const count = `${g.variants.length} variant${g.variants.length > 1 ? "s" : ""}`; - const detail = g.variants.map((v) => `${v.profile} (${describeDistance(v.distance)})`).join(", "); + const detail = g.variants.map((v) => `${v.profile} (${describeDistance(v.distance)})${describeClasses(v.classes)}`).join(", "); console.log(` ${g.base.padEnd(24)} ${count.padEnd(11)} ${detail}`); } for (const orphan of orphans) { @@ -226,7 +226,7 @@ forge for (const file of r.diff.files) { console.log(` ${file.file}`); file.hunks.forEach((h, k) => { - console.log(` hunk ${k + 1} [${h.kind}] ${hunkAt(h)}`); + console.log(` hunk ${k + 1} [${h.kind}] ${hunkAt(h)} ${describeSuggestion(h.suggestion)}`); for (const line of h.a.lines) console.log(pc.red(` - ${line}`)); if (h.a.noEofNewline) console.log(pc.red(` ${NO_EOF_NEWLINE_MARKER}`)); for (const line of h.b.lines) console.log(pc.green(` + ${line}`)); @@ -590,6 +590,18 @@ function lateFailure(e: unknown, root: string, journal: WriteJournal): string { return lines.join("\n"); } +/** ` [1 evolution · 2 block]` in the fixed class order, zero counts omitted; empty for a variant without hunks (spec 08 §4.2). */ +function describeClasses(classes: Record): string { + const parts = HUNK_CLASSES.filter((c) => classes[c] > 0).map((c) => `${classes[c]} ${c}`); + return parts.length ? ` [${parts.join(" · ")}]` : ""; +} + +/** `: `, and ` → ` for a value (spec 08 §4.1). */ +function describeSuggestion(s: HunkSuggestion): string { + const params = s.tokens?.length ? ` → ${s.tokens.map((t) => t.param).join(", ")}` : ""; + return `${s.class}: ${s.reason}${params}`; +} + function describeDistance(d: Distance): string { if (d.identicalAfterNormalization) return "identical after normalization"; if (d.sameBodyDifferentMeta) return "meta only"; diff --git a/test/cli.test.ts b/test/cli.test.ts index 630bc76..242cb14 100644 --- a/test/cli.test.ts +++ b/test/cli.test.ts @@ -1446,3 +1446,82 @@ describe("cli — forge unify --save-plan through a dangling symlink (Ruling 30, expect(await snapshot(root)).toEqual(before); }); }); + +describe("cli — suggested hunk classes (spec 08)", () => { + const BASE = "# W\nBump `package.json` before the PR.\nshared\nStep 6: open the PR.\nend\n"; + const VARIANT = "# W\nBump `Directory.Build.props` before the PR.\nshared\nStep 5: open the PR.\nend\nonly here\n"; + async function classForge(): Promise { + const root = await tmpDir("craftar-cli-forge-"); + cleanups.push(() => fs.rm(root, { recursive: true, force: true })); + await makeForge(root, { + ingredients: [ + rule("wf", BASE), + { meta: { type: "rule", name: "wf--acme", as: "wf" }, files: { "rule.md": VARIANT, "extra.md": "x\n" } }, + rule("m", "same\n"), + rule("m--acme", "same\n", { as: "m", targets: ["kiro"] }), + ], + }); + gitInit(root); + gitCommitAll(root, "init"); + return root; + } + + it("forge diff prints the class and reason after the unchanged prefix, and --json carries suggestion next to kind", async () => { + const root = await classForge(); + const text = runCli(["forge", "diff", "rule/wf", "--forge", root]); + expect(text.code, text.stderr).toBe(0); + expect(text.stdout).toContain("hunk 1 [inline] lines 2–2 value: 1 token differs → param.package_json"); + expect(text.stdout).toContain("hunk 2 [inline] lines 4–4 evolution: numbering differs"); + expect(text.stdout).toContain("block: only in the variant"); + const json = runCli(["forge", "diff", "rule/wf", "--forge", root, "--json"]); + const hunks = JSON.parse(json.stdout)[0].diff.files[0].hunks; + expect(hunks[0].kind).toBe("inline"); + expect(hunks[0].suggestion).toEqual({ class: "value", reason: "1 token differs", tokens: [{ a: "package.json", b: "Directory.Build.props", param: "param.package_json" }] }); + for (const h of hunks) expect("tokens" in h.suggestion).toBe(h.suggestion.class === "value"); + }); + + it("forge variants appends the class counts, none for a variant without hunks, and --json carries classes", async () => { + const root = await classForge(); + const text = runCli(["forge", "variants", "--forge", root]); + expect(text.code, text.stderr).toBe(0); + expect(text.stdout).toMatch(/acme \([^)]*\) \[1 evolution · 1 value · 2 block\]/); + expect(text.stdout).toMatch(/acme \(meta only\)(?! \[)/); + const { groups } = JSON.parse(runCli(["forge", "variants", "--forge", root, "--json"]).stdout); + const wf = groups.find((g: { base: string }) => g.base === "rule/wf").variants[0]; + expect(wf.classes).toEqual({ evolution: 1, value: 1, block: 2 }); + expect(wf.classes.evolution + wf.classes.value + wf.classes.block).toBe(wf.distance.hunks); + }); + + it("--save-plan writes a suggestion per hunk and none on one-sided entries; --plan ignores it", async () => { + const planDir = await tmpDir("craftar-cli-plan-"); + cleanups.push(() => fs.rm(planDir, { recursive: true, force: true })); + const root = await classForge(); + const planPath = path.join(planDir, "plan.yaml"); + expect(runCli(["forge", "unify", "rule/wf", "--profile", "acme", "--save-plan", planPath, "--forge", root]).code).toBe(0); + const plan = YAML.parse(await fs.readFile(planPath, "utf8")); + const paired = plan.files.find((f: { file: string }) => f.file === "rule.md"); + expect(paired.hunks.map((h: { suggestion: { class: string } }) => h.suggestion.class)).toEqual(["value", "evolution", "block"]); + expect(paired.hunks.every((h: { take: string }) => h.take === "keep")).toBe(true); + for (const f of plan.files.filter((f: { onlyIn?: string }) => f.onlyIn)) expect(f).not.toHaveProperty("suggestion"); + + for (const h of paired.hunks) h.take = "variant"; + for (const f of plan.files.filter((f: { onlyIn?: string }) => f.onlyIn)) f.take = "variant"; + const variants = { + edited: plan, + mangled: { ...plan, files: plan.files.map((f: { hunks?: object[] }) => (f.hunks ? { ...f, hunks: f.hunks.map((h) => ({ ...h, suggestion: { class: "bogus" } })) } : f)) }, + removed: { ...plan, files: plan.files.map((f: { hunks?: object[] }) => (f.hunks ? { ...f, hunks: f.hunks.map(({ suggestion: _, ...h }: { suggestion?: unknown }) => h) } : f)) }, + }; + const results: Record> = {}; + for (const [name, p] of Object.entries(variants)) { + const forgeRoot = await classForge(); + const file = path.join(planDir, `${name}.yaml`); + await fs.writeFile(file, YAML.stringify(p)); + const r = runCli(["forge", "unify", "rule/wf", "--profile", "acme", "--plan", file, "--forge", forgeRoot]); + expect(r.code, `${name}: ${r.stderr}`).toBe(0); + const snap = await snapshot(forgeRoot); + results[name] = Object.fromEntries(Object.entries(snap).filter(([k]) => !k.startsWith(".git/"))); + } + expect(results.mangled).toEqual(results.edited); + expect(results.removed).toEqual(results.edited); + }); +}); From 79d189a8afa5cc0a0eb729743d89abeae8e1c30b Mon Sep 17 00:00:00 2001 From: Luiz Lima Date: Fri, 25 Sep 2026 21:39:48 -0300 Subject: [PATCH 06/10] test(golden): pin the suggestion in the saved unify plan Nothing compared the committed plan with a fresh --save-plan; now it does, and the regenerated golden gains only the three suggestion lines. --- test/golden-unify.test.ts | 19 +++++++++++++++++++ test/golden/forge-unify-plan.yaml | 3 +++ 2 files changed, 22 insertions(+) diff --git a/test/golden-unify.test.ts b/test/golden-unify.test.ts index 4fffee5..81c906e 100644 --- a/test/golden-unify.test.ts +++ b/test/golden-unify.test.ts @@ -3,6 +3,7 @@ import { promises as fs } from "node:fs"; import { execFileSync } from "node:child_process"; import os from "node:os"; import path from "node:path"; +import YAML from "yaml"; import { listFiles } from "../src/core/forge.js"; import { runCli } from "./helpers/cli.js"; @@ -103,4 +104,22 @@ describe("golden: forge unify writes exact bytes into a Forge", () => { expect(got.equals(want), `${rel} differs`).toBe(true); } }); + + it("a fresh --save-plan over the input Forge reproduces the committed plan, suggestions included (spec 08 §9.4)", async () => { + const tmp = await fs.mkdtemp(path.join(os.tmpdir(), "craftar-golden-plan-")); + cleanups.push(() => fs.rm(tmp, { recursive: true, force: true })); + const forge = path.join(tmp, "forge"); + await copyTree(INPUT, forge); + gitInit(forge); + gitCommitAll(forge, "init"); + const planTmp = path.join(tmp, "plan.yaml"); + const save = runCli(["forge", "unify", "rule/workflow", "--profile", "acme", "--save-plan", planTmp, "--forge", forge]); + expect(save.code, save.stderr).toBe(0); + // The same three decisions test/helpers/regen-golden-unify.ts sets before writing the golden. + const plan = YAML.parse(await fs.readFile(planTmp, "utf8")); + plan.files.find((f: { file: string }) => f.file === "rule.md").hunks[0].take = "variant"; + plan.files.find((f: { file: string }) => f.file === "extra.md").take = "variant"; + plan.files.find((f: { file: string }) => f.file === "notes.md").take = "base"; + expect(YAML.stringify(plan)).toBe(await fs.readFile(PLAN, "utf8")); + }); }); diff --git a/test/golden/forge-unify-plan.yaml b/test/golden/forge-unify-plan.yaml index c3a3d13..e649cce 100644 --- a/test/golden/forge-unify-plan.yaml +++ b/test/golden/forge-unify-plan.yaml @@ -10,6 +10,9 @@ files: - hunk: 1 at: lines 4–4 take: variant + suggestion: + class: evolution + reason: prose differs - file: notes.md onlyIn: base take: base From 0a24b749a976e05fffd42eec611da835a4252208 Mon Sep 17 00:00:00 2001 From: Luiz Lima Date: Fri, 25 Sep 2026 21:40:21 -0300 Subject: [PATCH 07/10] docs(readme): describe hunk suggestions in the forge commands The README rows and the --help texts of forge variants, forge diff and --save-plan name the suggested class and say it never decides anything. --- README.md | 6 +++--- src/cli.ts | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/README.md b/README.md index a83ec70..fce4d78 100644 --- a/README.md +++ b/README.md @@ -57,9 +57,9 @@ From then on, change a rule in `forge/ingredients/rules//rule.md`, run `cr | `craftar diff [path]` | Line diff between disk and what the Forge would generate. | | `craftar explain ` | Which ingredient, recipe chain, target and origin produced a file. | | `craftar ls` | Recipes and ingredients resolved for this workspace. | -| `craftar forge variants [--forge \| --workspace ] [--json]` | Lists ingredients that have variants, nearest first, with the profile each came from and its distance to the base, then any variant whose base is missing from the Forge. Read-only; exits 0 either way. `--json` prints `{groups, orphans}`. | -| `craftar forge diff [--against ] [--forge \| --workspace ] [--json]` | Shows the differences between a base ingredient and each of its variants: a header carrying the same distance `forge variants` reports, then hunk by hunk, then the files that exist on only one side. Read-only. `--json` prints an array of `{ref, profile, distance, diff}`. | -| `craftar forge unify --profile

(--take base\|variant \| --plan \| --save-plan ) [--forge

\| --workspace ] [--json]` | Resolves one variant back into its base, hunk by hunk. `--save-plan` writes a reviewable plan with every decision set to `keep` and writes nothing else; it refuses a path that resolves inside the Forge (after `..` segments and symlinks) and a path that already exists, so it never overwrites a Forge file or a plan you already edited. `--plan` applies an edited plan, refusing when it is not for this exact ingredient/profile or either side's fingerprint moved since it was saved. `--take base\|variant` resolves every decision to that side. Writes the merged base and, once every difference is resolved, removes the variant and rewrites every recipe reference to it (`ingredients`) to name the base. It never deletes a recipe and never edits a profile or an `extends`: a `--` left identical to `` is reported as a warning, to be removed by hand after repointing the lists that name it — unify does not, because a workspace or an `extends` chain may also name `` and the recipe order or param precedence would change. Unify cannot reach workspaces, so a removed variant is reported as a warning for any `craftar.yaml` that disables it in `overrides.ingredients.disable`. Every recipe rewrite is checked before the first write, so one that cannot land (a reference behind a YAML alias) is refused with the Forge untouched; a failure after writing began (an I/O error, a locked file) names the paths already touched and the `git checkout` / `git clean` commands that undo them. `ingredient.yaml` is never merged: when the two sides' metadata differs (beyond `name`, `as` and `origin`), the variant stays unresolved and the differing fields are named — edit it by hand, or `--take base` to discard the variant, metadata included. Requires a clean git checkout in the Forge with at least one commit (`--save-plan` excepted) — the Forge has no lock, so git is the undo. It also refuses when any path it would overwrite or delete (the base, the variant, the recipe files it rewrites) is ignored, untracked, modified, or flagged skip-worktree or assume-unchanged in the index, since git could not restore it. `--json` prints `{base, profile, resolved, written, removed, unresolved, variantRemoved, recipes: {rewritten, identicalToSibling}, metaDiffers, warnings}` for `--take`/`--plan` (every key always present, `[]` when empty), or `{base, profile, plan, unresolved}` for `--save-plan`. | +| `craftar forge variants [--forge \| --workspace ] [--json]` | Lists ingredients that have variants, nearest first, with the profile each came from, its distance to the base and its hunks counted by suggested class (`[1 evolution · 2 block]`), then any variant whose base is missing from the Forge. Read-only; exits 0 either way. `--json` prints `{groups, orphans}`; each variant carries `classes: {evolution, value, block}`, which add up to `distance.hunks`. | +| `craftar forge diff [--against ] [--forge \| --workspace ] [--json]` | Shows the differences between a base ingredient and each of its variants: a header carrying the same distance `forge variants` reports, then hunk by hunk, each with a suggested class and reason — `evolution` (one side is newer text), `value` (an identifier-like token swapped in shared prose, with a suggested `param.`) or `block` (lines only one side has) — then the files that exist on only one side. A suggestion never decides anything. Read-only. `--json` prints an array of `{ref, profile, distance, diff}`; each hunk carries `suggestion: {class, reason, tokens?}` next to its `kind`. | +| `craftar forge unify --profile

(--take base\|variant \| --plan \| --save-plan ) [--forge

\| --workspace ] [--json]` | Resolves one variant back into its base, hunk by hunk. `--save-plan` writes a reviewable plan with every decision set to `keep`, each hunk annotated with its suggested class (which `--plan` ignores), and writes nothing else; it refuses a path that resolves inside the Forge (after `..` segments and symlinks) and a path that already exists, so it never overwrites a Forge file or a plan you already edited. `--plan` applies an edited plan, refusing when it is not for this exact ingredient/profile or either side's fingerprint moved since it was saved. `--take base\|variant` resolves every decision to that side. Writes the merged base and, once every difference is resolved, removes the variant and rewrites every recipe reference to it (`ingredients`) to name the base. It never deletes a recipe and never edits a profile or an `extends`: a `--` left identical to `` is reported as a warning, to be removed by hand after repointing the lists that name it — unify does not, because a workspace or an `extends` chain may also name `` and the recipe order or param precedence would change. Unify cannot reach workspaces, so a removed variant is reported as a warning for any `craftar.yaml` that disables it in `overrides.ingredients.disable`. Every recipe rewrite is checked before the first write, so one that cannot land (a reference behind a YAML alias) is refused with the Forge untouched; a failure after writing began (an I/O error, a locked file) names the paths already touched and the `git checkout` / `git clean` commands that undo them. `ingredient.yaml` is never merged: when the two sides' metadata differs (beyond `name`, `as` and `origin`), the variant stays unresolved and the differing fields are named — edit it by hand, or `--take base` to discard the variant, metadata included. Requires a clean git checkout in the Forge with at least one commit (`--save-plan` excepted) — the Forge has no lock, so git is the undo. It also refuses when any path it would overwrite or delete (the base, the variant, the recipe files it rewrites) is ignored, untracked, modified, or flagged skip-worktree or assume-unchanged in the index, since git could not restore it. `--json` prints `{base, profile, resolved, written, removed, unresolved, variantRemoved, recipes: {rewritten, identicalToSibling}, metaDiffers, warnings}` for `--take`/`--plan` (every key always present, `[]` when empty), or `{base, profile, plan, unresolved}` for `--save-plan`. | All commands take `--workspace ` (default: current directory); the `forge` commands also take `--forge ` as an alternative to it. diff --git a/src/cli.ts b/src/cli.ts index 8bd3f5e..ed2b3c0 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -164,7 +164,7 @@ const forge = program.command("forge").description("Operate on the Forge itself forge .command("variants") - .description("List ingredients that have variants, nearest first, and variants whose base is missing. Read-only") + .description("List ingredients that have variants, nearest first, with hunks counted by suggested class, and variants whose base is missing. Read-only") .option("--forge ", "Forge directory (instead of --workspace)") .option("--workspace ", "workspace whose craftar.yaml names the Forge (default: .)") .option("--json", "machine-readable output", false) @@ -192,7 +192,7 @@ forge forge .command("diff") - .description("Show the distance and the differences between a base ingredient and each of its variants. Read-only") + .description("Show the distance and the differences between a base ingredient and each of its variants, each hunk with a suggested class (evolution, value, block) that never decides anything. Read-only") .argument("", "base ingredient (rule/workflow)") .option("--against ", "only this profile's variant") .option("--forge ", "Forge directory (instead of --workspace)") @@ -245,7 +245,7 @@ forge .requiredOption("--profile

", "which variant to resolve") .option("--take ", "resolve every decision to base or variant") .option("--plan ", "apply the decisions in this plan file") - .option("--save-plan ", "write a plan with every decision deferred to a new file outside the Forge, and stop") + .option("--save-plan ", "write a plan with every decision deferred, each hunk annotated with its suggested class, to a new file outside the Forge, and stop") .option("--forge

", "Forge directory (instead of --workspace)") .option("--workspace ", "workspace whose craftar.yaml names the Forge (default: .)") .option("--json", "machine-readable output", false) From 1f996ac883eec26728f6ca65256543e80e50c9d9 Mon Sep 17 00:00:00 2001 From: Luiz Lima Date: Sat, 26 Sep 2026 07:08:23 -0300 Subject: [PATCH 08/10] fix(classify): never give two tokens of one hunk the same parameter name A suffixed name could equal another token's own slug (acme_api_2); the suffix now grows until the name is free. The contract test covers every suggestion the corpus produces. --- src/core/classify.ts | 11 ++++++---- test/classify.test.ts | 49 ++++++++++++++++++++++++------------------- 2 files changed, 35 insertions(+), 25 deletions(-) diff --git a/src/core/classify.ts b/src/core/classify.ts index 685e5ae..af6d646 100644 --- a/src/core/classify.ts +++ b/src/core/classify.ts @@ -105,16 +105,19 @@ export function classifyHunk(h: Hunk): HunkSuggestion { if (!changed.length) return evolution("whitespace only"); const seen = new Set(); - const slugCount = new Map(); + const used = new Set(); const tokens: NonNullable = []; for (const c of changed) { const id = JSON.stringify([c.a, c.b]); if (seen.has(id)) continue; seen.add(id); + // A suffixed name can equal another token's natural slug (`acme_api` → `_2` vs `acme_api_2`), + // so the suffix grows until the name is free: every distinct pair gets its own parameter. const slug = paramSlug(c.a); - const n = (slugCount.get(slug) ?? 0) + 1; - slugCount.set(slug, n); - tokens.push({ a: c.a, b: c.b, param: n === 1 ? slug : `${slug}_${n}` }); + let param = slug; + for (let n = 2; used.has(param); n++) param = `${slug}_${n}`; + used.add(param); + tokens.push({ a: c.a, b: c.b, param }); } return { class: "value", reason: tokens.length === 1 ? "1 token differs" : `${tokens.length} tokens differ`, tokens }; } diff --git a/test/classify.test.ts b/test/classify.test.ts index 966adb2..e50a4b4 100644 --- a/test/classify.test.ts +++ b/test/classify.test.ts @@ -9,7 +9,14 @@ function hunkOf(a: string, b: string) { expect(hunks).toHaveLength(1); return hunks[0]; } -const classify = (a: string, b: string) => classifyHunk(hunkOf(a, b)); +/** Every suggestion any test in this file produced, so the contract test covers the whole corpus (AC 2). */ +const produced: HunkSuggestion[] = []; +const classifyOne = (h: ReturnType) => { + const s = classifyHunk(h); + produced.push(s); + return s; +}; +const classify = (a: string, b: string) => classifyOne(hunkOf(a, b)); const brief = (s: HunkSuggestion) => `${s.class}: ${s.reason}`; const REASONS = new Set([ @@ -24,7 +31,7 @@ const REASONS = new Set([ "prose differs", "whitespace only", ]); -const allowed = (s: HunkSuggestion) => REASONS.has(s.reason) || /^(1 token differs|[0-9]+ tokens differ)$/.test(s.reason); +const allowed = (s: HunkSuggestion) => REASONS.has(s.reason) || /^(1 token differs|([2-9]|[1-9][0-9]+) tokens differ)$/.test(s.reason); describe("tokenize", () => { it("splits whitespace runs, words and single other characters", () => { @@ -120,7 +127,7 @@ describe("classifyHunk — edge cases (spec 08 §7)", () => { it("2. a line appended after an unterminated last line is a block, whatever the structural kind", () => { const h = hunkOf("a\nb", "a\nb\nc\n"); expect(h.kind).toBe("inline"); - expect(brief(classifyHunk(h))).toBe("block: only in the variant"); + expect(brief(classifyOne(h))).toBe("block: only in the variant"); }); it("3. a token swap plus a final-newline change is evolution", () => { @@ -143,6 +150,7 @@ describe("classifyHunk — edge cases (spec 08 §7)", () => { it("7. a changed row plus added rows is evolution; added rows alone are a block", () => { const base = "| a | acme-api |\n"; expect(brief(classify(base, "| a | globex-api |\n| b | globex-web |\n"))).toBe("evolution: line counts differ"); + expect(brief(classify(base, base + "| b | globex-web |\n"))).toBe("block: only in the variant"); }); it("8. known false positives: hyphenated words, abbreviations and years", () => { @@ -172,6 +180,17 @@ describe("classifyHunk — edge cases (spec 08 §7)", () => { expect(brief(classify(`Use a-${"x".repeat(39)} here.\n`, "Use acme-api here.\n"))).toBe("evolution: prose differs"); }); + it("11b. a suffixed name never collides with another token's own slug", () => { + const s = classify("acme-api x acme_api y acme_api_2\n", "p-1 x p-2 y p-3\n"); + expect(s.tokens?.map((t) => t.param)).toEqual(["param.acme_api", "param.acme_api_2", "param.acme_api_2_2"]); + }); + + it("13a. a line of exactly 400 tokens is still compared", () => { + const line = (w: string) => Array.from({ length: 200 }, () => w).join(" ") + "."; // 200 words, 199 spaces, "." + expect(tokenize(line("acme-api"))).toHaveLength(400); + expect(brief(classify(line("acme-api") + "\n", line("globex-api") + "\n"))).toBe("value: 1 token differs"); + }); + it("13. an over-long line is not compared word by word", () => { const long = (w: string) => Array.from({ length: 201 }, () => w).join(" ") + "\n"; expect(brief(classify(long("acme-api"), long("globex-api")))).toBe("evolution: line too long to compare"); @@ -204,24 +223,12 @@ describe("classifyHunk — whitespace and markdown (spec 08 §6.2 consequences)" }); describe("classifyHunk — contract (spec 08 AC 2)", () => { - it("returns only reasons from the closed set, and tokens only for a value", () => { - const corpus: Array<[string, string]> = [ - ["_TODO_: decide.\n", "See `x.md`.\n"], - ["Bump `package.json`.\n", "Bump `a.props`.\n"], - ["r\n", "r\nq\n"], - ["a\nb\n", "a\nb"], - ["Step 6.\n", "Step 5.\n"], - ["a\tb\n", "a b\n"], - ["x\nacme-api", "x\nglobex-api\n"], - ["one\n", "one\ntwo\nthree\n"], - ]; - for (const [a, b] of corpus) { - for (const h of diffLines(a, b)) { - const s = classifyHunk(h); - expect(allowed(s), s.reason).toBe(true); - expect(s.tokens !== undefined, s.reason).toBe(s.class === "value"); - for (const t of s.tokens ?? []) expect(t.param).toMatch(/^param\.[a-z0-9_]+$/); - } + it("every suggestion the corpus above produced uses a reason from the closed set, and tokens only for a value", () => { + expect(produced.length).toBeGreaterThan(30); + for (const s of produced) { + expect(allowed(s), s.reason).toBe(true); + expect(s.tokens !== undefined, s.reason).toBe(s.class === "value"); + for (const t of s.tokens ?? []) expect(t.param).toMatch(/^param\.[a-z0-9_]+$/); } }); }); From 85b8f9e5f68632ba5e2bfc7fac4493ecebee66d8 Mon Sep 17 00:00:00 2001 From: Luiz Lima Date: Sat, 26 Sep 2026 07:08:23 -0300 Subject: [PATCH 09/10] docs(cli): say in --save-plan's help that --plan ignores the suggestion The README row said so; the option text did not. --- src/cli.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/cli.ts b/src/cli.ts index ed2b3c0..fbbbf7b 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -245,7 +245,7 @@ forge .requiredOption("--profile

", "which variant to resolve") .option("--take ", "resolve every decision to base or variant") .option("--plan ", "apply the decisions in this plan file") - .option("--save-plan ", "write a plan with every decision deferred, each hunk annotated with its suggested class, to a new file outside the Forge, and stop") + .option("--save-plan ", "write a plan with every decision deferred, each hunk annotated with its suggested class (which --plan ignores), to a new file outside the Forge, and stop") .option("--forge

", "Forge directory (instead of --workspace)") .option("--workspace ", "workspace whose craftar.yaml names the Forge (default: .)") .option("--json", "machine-readable output", false) From 8d612d45bee1f88a8a58da2bb00bf2fee68e6ccf Mon Sep 17 00:00:00 2001 From: Luiz Lima Date: Sat, 26 Sep 2026 07:10:03 -0300 Subject: [PATCH 10/10] chore: update version --- package-lock.json | 4 ++-- package.json | 2 +- src/cli.ts | 2 +- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/package-lock.json b/package-lock.json index 934f7fb..ae12b8c 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "craftar", - "version": "0.3.0", + "version": "0.4.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "craftar", - "version": "0.3.0", + "version": "0.4.0", "license": "MIT", "dependencies": { "commander": "^13.1.0", diff --git a/package.json b/package.json index f624ffa..e65bb4b 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "craftar", - "version": "0.3.0", + "version": "0.4.0", "description": "Craft, sync and convert AI-coding workspace harnesses across clients and tools.", "license": "MIT", "type": "module", diff --git a/src/cli.ts b/src/cli.ts index fbbbf7b..7cb2eed 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -25,7 +25,7 @@ import { HUNK_CLASSES, UnifyPlanSchema, type HunkClass, type HunkSuggestion, typ process.stdout.on("error", (e: NodeJS.ErrnoException) => { if (e.code === "EPIPE") process.exit(0); }); const program = new Command(); -program.name("craftar").description("Craft, sync and convert AI-coding workspace harnesses.").version("0.3.0"); +program.name("craftar").description("Craft, sync and convert AI-coding workspace harnesses.").version("0.4.0"); /* ---------------------------------------------------------------- import */ program