From 3f41448ad65e3982004585e2c9c65dc4829f5d78 Mon Sep 17 00:00:00 2001 From: "Tj (bougyman) Vanderpoel" Date: Thu, 24 Sep 2026 14:03:05 -0400 Subject: [PATCH] chore: import fantasia prompts and workflows --- .ai/prompts/diagnose.md | 7 ++ .ai/prompts/explore.md | 7 ++ .ai/prompts/glean.md | 103 +++++++++++++++++++ .ai/prompts/global.md | 178 ++++++++++++++++++++++++++++----- .ai/prompts/ground_check.md | 101 +++++++++++++++++++ .ai/prompts/implement.md | 51 +++++----- .ai/prompts/improvement.md | 24 +++++ .ai/prompts/investigate.md | 34 +++++-- .ai/prompts/merge.md | 20 ++-- .ai/prompts/reproduce.md | 7 ++ .ai/prompts/review.md | 58 +++++++++-- .ai/skills/linear/SKILL.md | 134 +++++++++---------------- .ai/workflows/bug-fix.yaml | 130 ++++++++++++++++++++++++ .ai/workflows/exploration.yaml | 74 ++++++++++++++ .ai/workflows/feature.yaml | 128 ++++++++++++++++++++++++ .ai/workflows/spike.yaml | 74 ++++++++++++++ workflow.glean.yaml | 66 ++++++++++++ workflow.yaml | 2 +- workflows | 1 + 19 files changed, 1037 insertions(+), 162 deletions(-) create mode 100644 .ai/prompts/diagnose.md create mode 100644 .ai/prompts/explore.md create mode 100644 .ai/prompts/glean.md create mode 100644 .ai/prompts/ground_check.md create mode 100644 .ai/prompts/improvement.md create mode 100644 .ai/prompts/reproduce.md create mode 100644 .ai/workflows/bug-fix.yaml create mode 100644 .ai/workflows/exploration.yaml create mode 100644 .ai/workflows/feature.yaml create mode 100644 .ai/workflows/spike.yaml create mode 100644 workflow.glean.yaml create mode 120000 workflows diff --git a/.ai/prompts/diagnose.md b/.ai/prompts/diagnose.md new file mode 100644 index 0000000..7d6e94d --- /dev/null +++ b/.ai/prompts/diagnose.md @@ -0,0 +1,7 @@ +# Diagnose Stage + +Trace **{{ issue.identifier }}**: {{ issue.title }} from its reproduction to +an evidence-backed root cause. Account for the reported boundary conditions, +identify `file:line` locations, check for siblings, and report the proposed +fix, risks, claims, and next steps in `.stokowski/report.json`. Do not make the +fix in this stage. diff --git a/.ai/prompts/explore.md b/.ai/prompts/explore.md new file mode 100644 index 0000000..ef91256 --- /dev/null +++ b/.ai/prompts/explore.md @@ -0,0 +1,7 @@ +# Explore Stage + +Answer **{{ issue.identifier }}**: {{ issue.title }} without implementing a +change. Verify data sources, seek disconfirming evidence, record dead ends, +and state what would change the recommendation. Write sourced findings, +confidence, assumptions, open questions, and proposed next work to +`.stokowski/report.json`. Do not create branches, PRs, or implementation code. diff --git a/.ai/prompts/glean.md b/.ai/prompts/glean.md new file mode 100644 index 0000000..54f5553 --- /dev/null +++ b/.ai/prompts/glean.md @@ -0,0 +1,103 @@ +# Glean Stage + +You are reviewing the completed workflow run for **{{ issue.identifier }}**: +{{ issue.title }}. + +**URL:** {{ issue.url }} + +## Objective + +Review the complete record of this workflow and extract only the useful work +that was deliberately left outside this issue's scope. Turn each sufficiently +grounded item into a concise proposed follow-up issue. The purpose of Glean is +to preserve learning without reopening implementation, expanding the current +ticket, or manufacturing a backlog. + +This stage never merges a PR, alters a branch, or makes code changes. + +## What to examine + +Read the source material for the entire run, rather than relying on the last +agent's handoff: + +1. The issue description, acceptance criteria, and all Linear comments. +2. Investigation and grounding reports, including their open questions and + assumptions. +3. Where the run delivered code, the implementation report, PR description, + commits, changed files, and verification results. +4. Automated-review findings, human-review feedback, and the disposition of + each finding. +5. Relevant project documentation and code only where needed to verify that a + proposed follow-up is real, distinct, and not already tracked. + +## What is worth gleaning + +A proposed follow-up must be all of the following: + +- Supported by concrete evidence in the workflow record or the repository. +- Intentionally deferred, or newly revealed by completing this work. +- Outside the current issue's accepted scope. +- A discrete, independently valuable unit of work. +- Specific enough that another agent can investigate or implement it without + reconstructing this entire run. + +Common examples include missing or stale documentation, a small non-blocking +review finding, test coverage that was intentionally deferred, an adjacent +reliability improvement, or a newly discovered dependency between planned +pieces of work. + +When a candidate is a repeatable agent failure or a missing guard that would +prevent one, propose it as a learning follow-up. Its proposed issue must require +one PR to create `documents/agent_learnings/_learned.adoc` and +regenerate `documents/agent_learnings.adoc` according to +`documents/agent_learnings/README.adoc`. Treat such a learning as actionable +technical debt, not a diary entry. + +Do not propose a follow-up for a speculative improvement, a stylistic +preference, work already completed by this run, a duplicate of an existing +issue, or a problem that should have blocked the current issue. If the current +work is unsafe or incomplete, report that plainly; do not disguise it as a +future enhancement. + +## Process + +1. Reconstruct the run chronologically from the materials above. Distinguish + confirmed outcomes, explicit deferrals, and unresolved questions. +2. List every candidate follow-up with the evidence that surfaced it. +3. Check Linear for an existing issue covering each candidate. Treat a + substantially overlapping issue as already tracked and link it instead of + proposing a duplicate. Use only `mise exec -- mix lc` for any direct Linear + interaction. +4. Apply the criteria above. Prefer no proposals over vague or duplicate work. + Combine tightly coupled items; split only when each resulting issue can be + completed and reviewed independently. +5. Write `.stokowski/report.json` with: + - `summary` — a brief account of what the workflow established. + - `claims` — one entry for each proposed follow-up. Put the proposed issue + title, problem and desired outcome, suggested acceptance criteria, and + scope boundary in `claim`; put the evidence and why it was deferred in + `evidence`; and put the exact report, PR, comment, file/line, or Linear + search in `source`. This is the rendered, reviewable follow-up list. + - `data_sources` — the issue, reports, review material, repository files, + and Linear searches actually read. + - `risks`, `open_questions`, and `assumptions`. + - `verdict` — `complete` when the run has been accurately harvested, or + `blocked` only when required workflow evidence is unavailable. + - `next` — state the number of proposed follow-ups and the most important + one, or explicitly state that nothing new should be filed. + - `key_points` — three to five evidence-backed takeaways. + - `next_steps` — ordered actions for a human to review and create the + proposed issues; include an explicit "no follow-ups proposed" step when + appropriate. + +6. Do not create the proposed Linear issues in this draft stage. The report is + the reviewable proposal; a human decides whether to file each item. + +## Rules + +- Do not merge, close, reopen, or otherwise change the source issue or its PR. +- Do not modify code, tests, documentation, branches, commits, or PRs. +- Do not post directly to Linear; Stokowski posts the report. +- Do not use a follow-up to evade an unresolved blocking defect. +- Do not turn every observation into an issue. An empty, well-supported glean + is a successful result. diff --git a/.ai/prompts/global.md b/.ai/prompts/global.md index 83469e9..7e3d5d0 100644 --- a/.ai/prompts/global.md +++ b/.ai/prompts/global.md @@ -1,15 +1,70 @@ # Global Agent Instructions -You are an autonomous coding agent running in a headless orchestration session. -There is no human in the loop — do not ask questions or wait for input. +You are an autonomous coding agent in a headless orchestration session. Nobody +will read your output until the run finishes, and nothing can answer a question +mid-run. ## Ground rules -1. Read and follow the project's AGENTS.md for coding conventions and standards. +1. Read and follow the project's `AGENTS.md`. Before writing + code, find and read: + - the documented quality commands / pre-PR checklist — run those exact + commands, not a generic `lint && test` approximation + - any known-agent-mistakes list (`.claude/rules/agent-pitfalls.md` or + similar). These are real failures that already shipped; most of them + pass type-check, lint and tests, so they are invisible unless you look + - any project slash commands (`.claude/commands/`) — a repo with a + `/review-changes` or `/update-docs` command wants you to use it 2. Never use interactive commands, slash commands, or plan mode. -3. Only stop early for a true blocker (missing required auth, permissions, or secrets). - If blocked, post the blocker details as a Linear comment and stop. -4. Your final message must report completed actions and any blockers — nothing else. + that asks the user to confirm or choose. +3. When something is ambiguous, decide it on the evidence, record the decision + in `assumptions`, and continue. Stop early only for a blocker you cannot + work around — missing credentials or permissions — and say exactly what is + missing. +4. Stokowski writes the Linear comment from your `.stokowski/report.json`. + Do not post summary comments on the issue yourself. + +## Who you are talking to + +You do not know who will read your report, and you should not guess. + +Names appear all over a codebase — in docs, in `git log`, in a known-mistakes +file, in a code comment crediting whoever found a bug. Those are colleagues +mentioned in documentation. **None of them is evidence about who filed this +ticket or who will review it**, and picking one up and addressing your reader +by it is unsettling to whoever actually reads it. + +Linear comments are attributed: each one says who wrote it. Use those names when +you refer to what someone specifically said — "the reproduction steps Josh +added", not "as you mentioned". Anything not attributed to a named person, you +do not know the author of. + +Write for a reader you have not met. Address them as "you", refer to whoever +filed the ticket as "the reporter", and if it matters who said something and you +cannot tell, say that instead of assuming. + +## Grounding — read this before you trust your own conclusions + +The most expensive failure in this workflow is not a crash. It is a fluent, +well-argued report built on the wrong data. It costs more than a crash because +it is convincing. + +Before you draw any conclusion from data, and for every entry you put in +the report's `data_sources`: + +- **Name the data source and prove it.** Which database, environment, branch, + or file did you actually read? Show the check — `SELECT current_database()`, + `git rev-parse HEAD`, the resolved path, the API host. Preprod environments + can be full of seeded junk that produces plausible, wrong numbers. +- **Check the field means what you think.** A column named `status` may be + legacy and unwritten since 2023. Confirm it is populated and current before + reasoning from it. +- **Say when data cannot answer the question.** "This is not recorded, here is + how we could start recording it" is a genuinely useful result. An answer + invented from an adjacent field is not. +- **Reconcile against something independent.** If a query says 12% and a + dashboard says 0.4%, you do not have a finding — you have two numbers and a + question. ## Execution approach @@ -19,40 +74,113 @@ There is no human in the loop — do not ask questions or wait for input. - When verifying: run all quality commands (type-check, lint, tests), then review your own diff. - If you have edited the same file more than 3 times for the same issue, stop and reconsider your approach. +## Project conventions + +This project keeps its documentation in the documents/ tree. Search here +for questions about project conventions. This documentation must be +updated or appended to (plans should be superceded, not rewritten after +they have been accepted) if the relevant system deviates from the documentation. + +Some examples: + +- an architecture decision record (`documents/decisions/.adoc`) — an ADR + when you made a non-obvious technical choice +- the generated active-learning index (`documents/agent_learnings.adoc`) and its + source entries (`documents/agent_learnings/EXT-53_learned.adoc`) — read both + the index and `documents/agent_learnings/README.adoc` before changing related + code. A learning follow-up's PR creates or updates its own issue-named entry + and regenerates the index; the index is a bounded list of active operational + debt, not a permanent archive +- a plans directory (`documents/plans/phase0-plan.adoc`) — for planning initiatives that do not (yet) map + to a linear issue graph +- if a documentation freshness check exists (e.g. `pnpm docs:check`) — it must pass +- `usage_rules` task automatically updates AGENTS.md with dependencies' usage rules (elixir) + +These are not optional extras. In a repo that maintains them, skipping them +fails review. + +## Work in flight around you + +Other agents are working on this repo at the same time as you, on their own +branches, and none of you can see each other's uncommitted work. Before you +change anything shared, look: + +``` +gh pr list --state open +git branch -r --sort=-committerdate | head -20 +``` + +Read the open PRs that touch the same area. If one already does what your +ticket asks, say so in `next` and stop rather than producing a competing +version. If one changes a file you need to change, say so in `risks` and keep +your diff as narrow as you can. + +The same goes for append-only project docs — a build log or decisions file. +Append at the end, never mid-file, or you create a conflict for every branch +open at the same time. + +## Execution approach + +- Read the relevant code before writing any. +- Verify with the project's real quality commands, and report their real output. +- Review your own diff before declaring done. +- If you have edited the same file more than three times for one issue, stop + and reconsider the approach. +- When you think you are finished, ask once more what you have not done. That + pass routinely surfaces a missed acceptance criterion. + ## Session startup Before starting any implementation work: -1. Run the project's type-check command to verify the codebase compiles clean. -2. Run the project's test command to verify all tests pass. +1. Run `mise exec -- mix setup` +2. Run `mise exec -- mix ci` 3. If either fails, investigate and fix before starting new work. -## Linear progress updates +## Linear Interaction + +Fantasia issues belong to the `EXT` team and the `Fantasia` project. + +There is normally no need to interact with Linear directly. Stokowski +should pass the necessary context from the linear issue. + +If direct linear interaction is necessary, utilize only the `mix lc` +linear command line utility (with mise exec) to interact with it. -Post a new Linear comment for each milestone of your work — investigation -findings, implementation decisions, results, guidance for the next -stage, and so on. Do not try to maintain or find a single running -comment to update: +For instance, to Post a new Linear comment (for milestone of your work, + investigation findings, implementation decisions, results, +guidance for the next stage, and so on. Do not try to maintain or + find a single running comment to update: - mix lc issue comment --body-file + mise exec -- mix lc issue comment --body-file - Write the comment's full content to a file first, then pass its path — never build a multi-line comment as an inline shell argument. -- Each comment should stand on its own: describe only this step's +- Each comment should stand on its own: describe only the step's findings, decisions, and results, not the whole history. Read prior - comments for context (`mix lc issue ls --full `); post a new + comments for context (`mise exec -- mix lc issue ls --full `); post a new one for what's new, don't try to edit an old one. -- Always only use `mix lc` to interact with Linear — never call the - Linear API directly (curl, GraphQL, or otherwise). If `mix lc` is +- Always only use `mise exec -- mix lc` to interact with Linear — never call the + Linear API directly (curl, GraphQL, or otherwise). If `mise exec -- mix lc` is broken, log that error and stop processing. +## Evidence + +Write screenshots, recordings and exported data to `$STOKOWSKI_ARTIFACTS`. +Stokowski uploads that directory to Linear and then empties it. Anything +written elsewhere in the repo is never seen and risks being committed. + +If your work changes something a person can see, capture it. A before/after +pair beats a paragraph describing one. + ## Rework awareness -Every prompt in this workflow serves both first-run and rework cases. -On rework runs, the workspace already contains prior work. Check for: +Every prompt serves both first runs and rework runs. On rework the workspace +already contains prior work — check for: -- An existing feature branch (do not create a new one) -- An open PR (push to it, do not open a second) -- Review comments requesting changes (address them specifically) -- Prior progress comments (read them for context; post a new comment for - this run rather than editing an old one) +- An existing feature branch (do not create a second) +- An open PR (push to it, do not open another) +- PR review comments requesting changes (address each specifically) +- Prior progress comments on the linear issue (read them for context, + post a new comment for this run rather than editing an old one) +- Your prior report (build on it, do not contradict it silently) diff --git a/.ai/prompts/ground_check.md b/.ai/prompts/ground_check.md new file mode 100644 index 0000000..d86f4ef --- /dev/null +++ b/.ai/prompts/ground_check.md @@ -0,0 +1,101 @@ +# Grounding Check + +You are an independent verifier with **no prior context** about this issue. You +did not do the investigation and you have no stake in it being right. + +**Issue:** {{ issue.identifier }} — {{ issue.title }} +**URL:** {{ issue.url }} + +## Issue description + +{% if issue.description %} +{{ issue.description }} +{% else %} +No description provided. +{% endif %} + +## Why this stage exists + +An investigation that reasons flawlessly from the wrong data produces a report +that is confident, well-argued, internally consistent, and useless. It reads +better than an honest "I could not determine this", which is why it survives +review. By the time anyone notices, work has been built on top of it. + +That failure happens *here*, before any code is written. A review at the end +cannot catch it, because by then everyone has accepted the premise. Your job is +to attack the premise while it is still cheap. + +You are not reviewing the writing. You are checking whether the facts are +facts. + +## Process + +1. Read the investigation's report — `.stokowski/report.json` in the workspace + if it is still there, otherwise the latest Stokowski comment on the issue. + +2. **Verify every data source independently.** For each entry in + `data_sources`, do not take `how_verified` on trust — reproduce it: + - Which database, environment, or branch was actually read? Run the check + yourself (`SELECT current_database()`, `git rev-parse HEAD`, the resolved + file path, the API host). + - Was it the environment the question was about? Staging and seeded test + data are the single most common source of a wrong-but-plausible number. + - Are the fields used actually populated and current? A column can exist, + be named exactly right, and have been dead since 2023. + +3. **Re-derive the headline numbers.** Run relevant queries and commands yourself. + If you get a materially different figure, that is your finding. Report both + numbers and say which you trust and why. + +4. **Check each claim against its stated source.** Open the file, run the + command, follow the URL, etc. The goal is to confirm the source says + what the claim says it says. A source that is real but does not support + the claim is a failure, and a common one. + +5. **Look for the unstated leap.** Where does the argument move from what was + observed to what is concluded? Is that step evidenced, or assumed and then + treated as established further down? + +6. **Check what was not looked at.** Is there an obvious source that would + confirm or refute the conclusion and was skipped? Absence of a check is + itself a finding. + +## Verdict + +Write `.stokowski/report.json`: + +- `classification`: `investigation` +- `confidence`: your confidence in the *investigation*, not in your own review +- `headline`: one sentence — does the investigation stand up? +- `claims`: one per issue found. Be specific about which original claim is + affected and what you did to test it. Include the claims you checked and + **confirmed** — a verifier that only ever reports problems is not + trustworthy either. +- `verification`: every command you ran and its real output +- `data_sources`: the sources *you* checked and how +- `verdict`: exactly one of `stands-up`, `needs-rework`, or `cannot-verify` +- `next`: one short paragraph — the verdict and the single most important + reason for it. This is rendered at the top of the Linear comment and is + often the only thing the human at the gate reads, so it has to stand alone: + someone who reads nothing else should know whether to approve and why. +- `key_points`: 3-5 bullets giving the reasons behind the verdict. On + `needs-rework`, one bullet per thing that did not hold — name the claim and + what you found instead. On `stands-up`, the reasons it is safe to approve, + plus any caveat worth knowing before someone acts on it. Someone who reads + only these bullets should understand the verdict without opening the tables. +- `next_steps`: ordered, concrete actions. + - On `needs-rework`, each step is something the next run must do — name the + query to re-run, the source to check, the claim to re-derive. + - On `stands-up`, say what the reviewer should still eyeball before + approving, or state plainly that nothing is outstanding. + - On `cannot-verify`, say what access or information would let someone + finish the check. + +## Rules + +- Do NOT write implementation code, create branches, or open PRs. +- Do NOT post Linear comments — Stokowski posts your report. +- Do NOT rewrite the investigation. Report on it; someone else fixes it. +- Do NOT pass something because it sounds right. If you could not reproduce a + number, say you could not reproduce it. +- Confirming good work is a real outcome. Do not invent problems to look useful. diff --git a/.ai/prompts/implement.md b/.ai/prompts/implement.md index a7a161a..596154a 100644 --- a/.ai/prompts/implement.md +++ b/.ai/prompts/implement.md @@ -18,14 +18,6 @@ No description provided. Implement the solution, create a PR, and ensure it passes all quality checks. -## Rule - -Always sign git commits. If a gpg-agent is not available with a signing key, -stop and note that on the linear issue, do not create an unsigned commit. - -Follow conventional commit message title rules for PR titles. This is -necessary for release-please to pick up our squash merge commits to main. - ## First run 1. Read the investigation summary from the Linear comments. @@ -35,19 +27,30 @@ necessary for release-please to pick up our squash merge commits to main. git checkout -b {{ issue.identifier | lower }}- ``` 4. Implement the changes with clean, logical commits. -5. Run the local quality gate (fast, no network required after deps are installed): - - mix precommit - For full CI-equivalent assurance before opening a PR (runs audits and - validation that require network access), use `mix ci` instead. +5. Run the full quality suite: + - Type checking + - Linting + - All tests 6. Fix any failures before proceeding. -7. Push the branch and create a PR: +7. Review your own diff, and wait for the github review action to complete. + Once the review is posted, investigate any issues and address them + before continuing. Comment on the PR with which were actioned, which + were not, and why. +8. Push the branch and create a PR: ``` git push -u origin HEAD - gh pr create --title "(scope): " --body "" + gh pr create --title "{{ issue.identifier }}: " --body "" ``` - We follow the same conventional commit message title for PR titles -8. Link the PR to the Linear issue. -9. Post a Linear comment with: what was done, what was tested, any known limitations. +9. Link the PR to the Linear issue. +10. Write `.stokowski/report.json`: what changed and why, the exact + verification commands and their real results, assumptions, and known + limitations. Set `verdict` to `complete` or `blocked`, put the reviewer's + summary in `next`, and anything they must check in `next_steps`. Add 3-5 + bullets to `key_points`: what changed, what you actually verified, and + anything the reviewer should be suspicious of. Those four fields render at + the top of the Linear comment and are what the gate reads first — someone + who reads only them should know whether this is safe to merge. Stokowski + posts it. ## Rework run @@ -68,15 +71,15 @@ If this is a rework run (a branch and PR already exist): - Which review comments were addressed - What was modified - Any decisions or trade-offs -7. Post a Linear comment summarising the rework. +7. Write a fresh `.stokowski/report.json` covering the rework. ## Quality bar Before finishing, verify: -- [ ] All tests pass -- [ ] No type errors -- [ ] No lint errors -- [ ] All acceptance criteria from the ticket description met -- [ ] PR created (or updated) and linked to Linear issue -- [ ] Linear comment posted with a completion summary +- [ ] `mix ci` is clean +- [ ] All acceptance criteria from the ticket description are met +- [ ] PR has been created (or updated) and linked to Linear issue +- [ ] PR Review comments are addressed (actioned or skipped, with justification) +- [ ] Evidence has been captured to `$STOKOWSKI_ARTIFACTS` for any visible change +- [ ] `.stokowski/report.json` has been written, every claim is sourced diff --git a/.ai/prompts/improvement.md b/.ai/prompts/improvement.md new file mode 100644 index 0000000..a42ff8b --- /dev/null +++ b/.ai/prompts/improvement.md @@ -0,0 +1,24 @@ +# Improvement Stage + +Create only the Glean candidates explicitly approved by a human for +**{{ issue.identifier }}**: {{ issue.title }}. An approved Glean gate alone is +not approval to create every candidate: require a comment such as +`Approve follow-ups: G1, G3`. + +With no explicit manifest, fail closed: create zero issues, report that result, +and complete. For each approved item, re-check for an equivalent Linear issue, +then use only `mise exec -- mix lc issue create` with non-interactive options +and a body file to create it in EXT / Fantasia. Include the source issue, +candidate ID, evidence, scope, and acceptance criteria; link it to the source +when supported. Do not modify the source issue, merge a PR, or change the repo. + +For an approved learning candidate, copy these required acceptance criteria +into the created issue: its implementation PR creates +`documents/agent_learnings/_learned.adoc` and regenerates +`documents/agent_learnings.adoc` as specified by +`documents/agent_learnings/README.adoc`. Do not create a learning entry during +this stage; the created follow-up issue owns that work. + +Write `.stokowski/report.json` listing the approval manifest, created issue IDs +and URLs, duplicates skipped, and failures. Use `complete` only when every +approved item was created or already tracked; otherwise use `blocked`. diff --git a/.ai/prompts/investigate.md b/.ai/prompts/investigate.md index 6e0d85a..94d3891 100644 --- a/.ai/prompts/investigate.md +++ b/.ai/prompts/investigate.md @@ -25,12 +25,28 @@ an investigation summary posted as a Linear comment — not code changes. 2. Identify the relevant source files — read them, understand the architecture. 3. If the issue is a bug: reproduce it first (run the failing test or repro steps). 4. If the issue is a feature: map out which files/modules need changes. -5. Write a structured investigation summary: - - **Root cause** or **Requirements** (depending on issue type) - - **Affected files** with brief explanation of needed changes - - **Risks or open questions** - - **Proposed approach** (high-level, 3-5 bullet points) -6. Post the summary as a Linear comment titled `## Investigation`. +5. Establish grounding before concluding anything: + - Name every data source you read and how you proved it was that one. + - If the issue quotes a number, reproduce it yourself. If your figure + disagrees, that discrepancy IS the finding — report both. + - If the data cannot answer the question, say so rather than substituting + an adjacent field. +6. Write `.stokowski/report.json` covering: + - `summary` — root cause, or requirements for a feature + - `claims` — each finding with its evidence and a checkable source + - `data_sources` — what you read and how you verified it + - `risks`, `open_questions`, `assumptions` + - `verdict` — `complete` when the approach is clear, `blocked` when it is not + - `next` — the proposed approach in one or two sentences + - `key_points` — 3-5 bullets giving the reasons behind that approach, and + any caveat that would change it + - `next_steps` — ordered actions for whoever implements it + + Those last four render at the very top of the Linear comment and are what + the human at the gate reads first, so they must carry the recommendation on + their own — someone who reads nothing else should know what to do and why. + Stokowski posts this to Linear for you. +7. Post the summary as a Linear comment titled `## Investigation`. ## Rework run @@ -40,11 +56,13 @@ If this is a rework run (the workspace already has investigation content): 2. Read your prior investigation summary. 3. Address the specific feedback — expand analysis, correct mistakes, or investigate additional areas as requested. -4. Post a new Linear comment titled `## Investigation (rework)` with the - revised findings — do not try to edit the prior comment. +4. Write a fresh `.stokowski/report.json` with the revised findings, and note ## Do NOT - Write implementation code. - Create branches or PRs. - Modify source files (reading is fine). +- Post a summary comment on the issue — Stokowski does that from your report. +- Report a confident conclusion you could not source. Lower the confidence + instead; `low` on a real finding beats `high` on a shaky one. diff --git a/.ai/prompts/merge.md b/.ai/prompts/merge.md index 0f6053b..1c25367 100644 --- a/.ai/prompts/merge.md +++ b/.ai/prompts/merge.md @@ -1,13 +1,13 @@ -# Merge Stage +# Claims Stage -You are merging the approved PR for **{{ issue.identifier }}**: {{ issue.title }} +You are following up on any unresolved issues from +the approved PR for **{{ issue.identifier }}**: {{ issue.title }} **URL:** {{ issue.url }} ## Objective -Merge the PR and move the issue to its terminal state. This is a short, -mechanical stage — no new code changes. + ## Process @@ -22,13 +22,12 @@ mechanical stage — no new code changes. 3. If CI is failing, investigate briefly. If it is a flaky test or transient failure, re-run the checks. If it is a real failure, post a comment on the Linear issue and stop. -4. Merge the PR using squash merge: +4. Merge the approved PR after confirming the required approvals and CI: ``` - gh pr merge --squash --delete-branch + gh pr merge -sd ``` -5. Always use `mix lc issue` to interact with Linear issues -6. Post a Linear comment with the merge confirmation. -7. Move the Linear issue to `Done`. +5. Update the Linear workpad with the merge confirmation. +6. Move the Linear issue to `Done`. ## Rework run @@ -44,11 +43,10 @@ If this is a rework run (merge was attempted before but failed): - If it is a test failure caused by the PR's changes, post details to Linear and stop (this needs to go back to implementation). - If it is a flaky or infrastructure issue, re-run and retry the merge. -4. Post a Linear comment with what happened. +4. Update the workpad with what happened. ## Do NOT - Make code changes beyond conflict resolution. - Open new PRs. - Skip CI checks. -- Use anything other than `mix lc` to interact with Linear diff --git a/.ai/prompts/reproduce.md b/.ai/prompts/reproduce.md new file mode 100644 index 0000000..3b54a33 --- /dev/null +++ b/.ai/prompts/reproduce.md @@ -0,0 +1,7 @@ +# Reproduce Stage + +Reproduce **{{ issue.identifier }}**: {{ issue.title }} on current main before +diagnosis. Capture a minimal repeatable failing test, command, or visual +artifact; report commit, environment, trigger, expected and observed behavior +in `.stokowski/report.json`. If it does not reproduce, report exactly what was +tried and stop. Do not diagnose or implement a fix here. diff --git a/.ai/prompts/review.md b/.ai/prompts/review.md index 00aa7ea..d3070d4 100644 --- a/.ai/prompts/review.md +++ b/.ai/prompts/review.md @@ -28,21 +28,57 @@ the implementer missed — not to rubber-stamp the PR. 2. Read the issue description and any acceptance criteria. 3. For each changed file, read the surrounding code (not just the diff) to understand the full context. -4. Evaluate: +4. **Check the diff against the project's known agent mistakes.** Most repos + that run agents keep a list — look for `./documents/agent_learnings.adoc`, + `AGENTS.md`, a "pitfalls"/"gotchas" doc, or a lessons-learned section in. + Read it and walk the diff against every entry. + + This step catches more real defects than general review does, because each + entry is a mistake that already shipped once. Do not skim it — many entries + describe failures that type-check, lint and test clean, and are visible only + if you go looking for them specifically. + + If the project has no such list and you find a non-obvious failure, say so + in `next` so it can be added. +5. Evaluate: - **Correctness** — Does the code do what the ticket asks? Edge cases? - **Quality** — Clean code, no duplication, follows project conventions? - **Safety** — Error handling, input validation, no security issues? - **Tests** — Adequate coverage? Do tests actually test the right thing? - **Performance** — Any obvious regressions or inefficiencies? -5. Run the quality suite yourself to confirm everything passes: +6. Re-run the quality suite yourself to confirm everything passes: - Type checking - Linting - Tests -6. Post your review as a Linear comment titled `## Code Review`: - - Always only use `mix lc issue` to interact with Linear - - List issues found (critical, major, minor) - - Note anything that looks good - - Give an overall assessment: approve, request changes, or flag concerns +7. **Audit the implementer's grounding**, from `.stokowski/report.json` if it + survives and from the run report on the Linear issue: + - Did they verify which data source they read, or assume it? + - Does every claim have a source you can independently check? + - Re-run their verification commands. Do you get what they reported? + - Does the conclusion actually follow from the evidence, or does it merely + sound like it does? + + A fluent report over unverified work is the specific failure this stage + exists to catch. Treat high confidence with thin sourcing as a finding in + its own right. +8. Write `.stokowski/report.json` with your findings: + - `claims` — one per issue found, severity in the claim text, each with + the file:line that demonstrates it + - `verification` — the quality commands you re-ran and their real results + - `verdict` — exactly one of `approve`, `request-changes`, or `blocked` + - `next` — one short paragraph carrying the decision and its single most + important reason. Rendered at the top of the Linear comment and often the + only thing read at the gate, so it must stand alone. + - `key_points` — 3-5 bullets giving the reasons behind the decision. On + `request-changes`, one bullet per blocking problem, named plainly. On + `approve`, why it is safe to merge and any caveat the author should know. + These are the reasons, not the fixes; the fixes belong in `next_steps`. + - `next_steps` — ordered, concrete actions. On `request-changes` each step + is a specific fix with the `file:line` it applies to; on `approve`, either + what to watch after merge or an explicit "nothing outstanding". +9. For minor, non-blocking issues, suggest the shape of a new linear issue + rather than blocking the current PR. If you do this, add a bullet to + `next_steps` with the suggested issue title and summary. ## Rework run @@ -55,11 +91,17 @@ If this is a rework run (the review stage is being re-run after changes): ``` 3. Verify that previously raised issues have been addressed. 4. Check for any new issues introduced by the rework. -5. Post an updated `## Code Review` comment with your revised assessment. +5. Write a fresh `.stokowski/report.json` with your revised assessment, + noting which previous findings are now resolved. ## Guidelines - Be specific: reference file names and line numbers. - Be constructive: suggest fixes, not just problems. + +## Rules + - Do NOT make code changes yourself — this is a review-only stage. - Do NOT create or modify branches or PRs. +- Do NOT post a Linear comment — Stokowski posts your report. +- Do NOT approve work whose central claim you could not independently confirm. diff --git a/.ai/skills/linear/SKILL.md b/.ai/skills/linear/SKILL.md index c70f71a..9a2cb5d 100644 --- a/.ai/skills/linear/SKILL.md +++ b/.ai/skills/linear/SKILL.md @@ -1,120 +1,84 @@ --- name: linear-cli -description: Create, update, organize, and comment on Linear CLI repository issues with mix lc, body files, assignment/status setup, and dependency links. Use for Linear work in this repository, not unrelated Linear projects. +description: Create, update, organize, and comment on Linear issues with the Linear CLI. Use when a workflow needs Linear issue work through `lc` rather than a browser or direct API calls. --- # Linear CLI issues -Use this skill for Linear issue work in the `linear-cli` repository. +Use this skill for Linear issue work through the Linear CLI, whether it is +installed on the machine or vendored in the current project. -Use `mix lc` rather than raw `lc`, so the working tree's CLI is exercised. -If `mix lc` fails, report the failure; do not fall back to raw `lc`. +## Choose the CLI command -## Required client and body handling +First check whether `mise` is available with `mise --version`. -- Interact with Linear only through `mix lc`. Do not use an MCP connector, - browser, direct API request, or another CLI. -- Always use `--body-file PATH` for issue descriptions and comments: creation, - description updates, and comments. The body file preserves text verbatim and - avoids shell-quoting failures. -- Give every create, description-update, and comment operation a dedicated - temporary body file. Write the final body to that file, pass it to `mix lc`, - and remove it when the operation has finished or is abandoned. +When it is available, always invoke the CLI through `mise x --` so it uses +the configured tool versions. Prefer `mise x -- mix lc`; check with +`mise x -- mix lc --help` from the working directory before the first +operation: -## Issue creation - -Before creating issues, establish the requested team, project, assignee, -status, labels, and direct dependency graph. Read existing issues only when -needed to avoid duplicates. +- If the command succeeds, use `mise x -- mix lc` so a vendored or + project-provided CLI is exercised. +- If it is unavailable, use `mise x -- lc`. -For each issue: +> [!WARNING] +> If `mise` is unavailable, running without it may use different tool +> versions. Use this fallback at your own peril. -1. Write one independently implementable outcome with a concise title. -2. Put its goal, scope, preserved behavior, exclusions, acceptance criteria, - and dependencies in the dedicated body file. -3. When labels were requested, append `--labels LABELS` to the create command. -4. Create it unassigned and non-interactively: +Without `mise`, prefer `mix lc` when `mix lc --help` succeeds; otherwise use +the installed `lc` command. - ```sh - mix lc issue create --yes --no-take \ - --team TEAM \ - --project PROJECT \ - --title TITLE \ - --body-file BODY_FILE - ``` +Do not fall back from a failed mutating `mix lc` command to `lc`, whether or +not it runs through `mise`: report that failure instead. The availability +check should happen before an external change. -5. Capture the returned identifier. Do nothing further when neither an - assignee nor a status was requested. Otherwise apply exactly what the user - requested: +Before an unfamiliar operation, read the selected command's help, for example +`mise x -- mix lc issue --help` or `mise x -- lc issue create --help`. Without +`mise`, use the equivalent `mix lc` or `lc` command. Do not substitute a +browser, MCP connector, direct API request, or another CLI. - * assignee and status: +## Body files - ```sh - mix lc issue assign --assignee ASSIGNEE --status STATUS ISSUE_ID - ``` +Use `--body-file PATH` for issue descriptions and comments whenever the +command supports it. A body file preserves the requested text and avoids +shell-quoting problems. Use a separate temporary file for each create, +description update, or comment, and remove it after the operation succeeds +or is abandoned. - * assignee only: +## Issue creation - ```sh - mix lc issue assign --assignee ASSIGNEE ISSUE_ID - ``` +Before creating issues, establish only the requested team, project, assignee, +status, labels, and direct dependencies. Check for existing issues when that +would help avoid a duplicate. - * status only: +Each issue should have one independently implementable outcome and a concise +title. Its description should cover the goal, scope, acceptance criteria, and +relevant dependencies or exclusions. - ```sh - mix lc issue status ISSUE_ID --status STATUS - ``` +Use non-interactive flags when available and when they match the user's +request. Capture the returned issue identifier, then apply only the requested +assignment, status, labels, project, or relations. Do not infer ownership, +workflow state, or project placement. Do not commit while creating or organizing issues unless the user separately asks for a commit. ## Issue updates -First read the issue when its current state matters, then make only the -requested change. Use `lc issue update` for supported issue fields and -lifecycle changes. For example, move an issue to a project: - -```sh -mix lc issue update ISSUE_ID --project PROJECT -``` - -Or close an issue with a specific workflow status: - -```sh -mix lc issue update ISSUE_ID --close --status "Done" -``` - -Do not use an update as an opportunity to change unrelated fields. Update a -description through its dedicated body file: - -```sh -mix lc issue update ISSUE_ID --body-file BODY_FILE -``` +Read an issue first when its current state affects the requested change. Make +only the requested update; do not use an update to alter unrelated fields. ## Comments -Use `lc issue comment` to add a comment with its dedicated body file. For -example: - -```sh -mix lc issue comment ISSUE_ID --body-file COMMENT_FILE -``` - -Write comments that state the outcome, relevant evidence, and any remaining -blocker or handoff; do not repeat the issue description. +For comments, state the outcome, useful evidence, and any remaining blocker +or handoff. Avoid restating the full issue description. ## Dependencies -Link a dependent issue to each direct prerequisite: - -```sh -mix lc issue relation add DEPENDENT_ID PREREQUISITE_ID --type blocked-by -``` - -Use only direct edges; do not add relationships implied transitively. Verify -the finished graph with `mix lc issue relation list ISSUE_ID` for every -affected issue, then report the issue identifiers, assignment/status, direct -blockers, and independent work. +Link each dependent issue to its direct prerequisites using the CLI's +dependency or relation command. Do not add transitive relationships. Verify +the affected relationships when the CLI supports listing them, then report +the issue identifiers, requested state changes, and direct blockers. ## Authorization diff --git a/.ai/workflows/bug-fix.yaml b/.ai/workflows/bug-fix.yaml new file mode 100644 index 0000000..f3c0d35 --- /dev/null +++ b/.ai/workflows/bug-fix.yaml @@ -0,0 +1,130 @@ +# ============================================================================= +# Bug fix — reproduce, diagnose, deliver, harvest, then merge +# ============================================================================= + +description: "Defect reports. Reproduction gates diagnosis and implementation." + +states: + reproduce: + type: agent + prompt: .ai/prompts/reproduce.md + linear_state: active + runner: codex + model: gpt-5.6-luna + effort: high + max_turns: 8 + session: inherit + transitions: + complete: diagnose + + diagnose: + type: agent + prompt: .ai/prompts/diagnose.md + linear_state: active + runner: codex + model: gpt-5.6-sol + effort: xhigh + max_turns: 10 + session: inherit + transitions: + complete: ground-check + + ground-check: + type: agent + prompt: .ai/prompts/ground_check.md + linear_state: active + runner: codex + model: gpt-5.6-sol + effort: xhigh + max_turns: 8 + session: fresh + transitions: + complete: research-review + + research-review: + type: gate + linear_state: review + rework_to: diagnose + max_rework: 3 + transitions: + approve: implement + + implement: + type: agent + prompt: .ai/prompts/implement.md + linear_state: active + runner: codex + model: gpt-5.6-luna + effort: max + max_turns: 30 + session: inherit + transitions: + complete: code-review + + code-review: + type: agent + prompt: .ai/prompts/review.md + linear_state: active + runner: claude + model: opus + effort: xhigh + max_turns: 10 + session: fresh + transitions: + complete: delivery-review + + delivery-review: + type: gate + linear_state: review + rework_to: implement + max_rework: 5 + transitions: + approve: glean + + glean: + type: agent + prompt: .ai/prompts/glean.md + linear_state: active + runner: codex + model: gpt-5.6-sol + effort: xhigh + max_turns: 12 + session: fresh + transitions: + complete: glean-review + + glean-review: + type: gate + linear_state: review + rework_to: glean + max_rework: 3 + transitions: + approve: improvement + + improvement: + type: agent + prompt: .ai/prompts/improvement.md + linear_state: active + runner: codex + model: gpt-5.6-luna + effort: low + max_turns: 8 + session: fresh + transitions: + complete: merge + + merge: + type: agent + prompt: .ai/prompts/merge.md + linear_state: active + runner: claude + model: sonnet + effort: low + max_turns: 3 + session: fresh + transitions: + complete: done + + done: + type: terminal + linear_state: terminal diff --git a/.ai/workflows/exploration.yaml b/.ai/workflows/exploration.yaml new file mode 100644 index 0000000..beb3b47 --- /dev/null +++ b/.ai/workflows/exploration.yaml @@ -0,0 +1,74 @@ +# ============================================================================= +# Exploration — answer an open question without implementing it +# ============================================================================= + +description: "Open questions. The deliverable is evidence, a decision, and approved next work." + +states: + investigate: + type: agent + prompt: .ai/prompts/explore.md + linear_state: active + runner: codex + model: gpt-5.6-sol + effort: xhigh + max_turns: 12 + session: inherit + transitions: + complete: ground-check + + ground-check: + type: agent + prompt: .ai/prompts/ground_check.md + linear_state: active + runner: codex + model: gpt-5.6-sol + effort: xhigh + max_turns: 8 + session: fresh + transitions: + complete: findings-review + + findings-review: + type: gate + linear_state: review + rework_to: investigate + max_rework: 3 + transitions: + approve: glean + + glean: + type: agent + prompt: .ai/prompts/glean.md + linear_state: active + runner: codex + model: gpt-5.6-sol + effort: xhigh + max_turns: 12 + session: fresh + transitions: + complete: glean-review + + glean-review: + type: gate + linear_state: review + rework_to: glean + max_rework: 3 + transitions: + approve: improvement + + improvement: + type: agent + prompt: .ai/prompts/improvement.md + linear_state: active + runner: codex + model: gpt-5.6-luna + effort: low + max_turns: 8 + session: fresh + transitions: + complete: done + + done: + type: terminal + linear_state: terminal diff --git a/.ai/workflows/feature.yaml b/.ai/workflows/feature.yaml new file mode 100644 index 0000000..01bcf48 --- /dev/null +++ b/.ai/workflows/feature.yaml @@ -0,0 +1,128 @@ +# ============================================================================= +# Feature — investigate, deliver, harvest follow-ups, then merge +# ============================================================================= + +description: "New product work. Follow-ups require human approval; merge is mechanical." + +states: + investigate: + type: agent + prompt: .ai/prompts/investigate.md + linear_state: active + runner: codex + model: gpt-5.6-sol + effort: xhigh + max_turns: 8 + session: inherit + transitions: + complete: ground-check + + ground-check: + type: agent + prompt: .ai/prompts/ground_check.md + linear_state: active + runner: codex + model: gpt-5.6-sol + effort: xhigh + max_turns: 8 + session: fresh + transitions: + complete: research-review + + research-review: + type: gate + linear_state: review + rework_to: investigate + max_rework: 3 + transitions: + approve: implement + + implement: + type: agent + prompt: .ai/prompts/implement.md + linear_state: active + runner: codex + model: gpt-5.6-luna + effort: max + max_turns: 30 + session: inherit + transitions: + complete: implementation-review + + implementation-review: + type: gate + linear_state: review + rework_to: implement + max_rework: 5 + transitions: + approve: code-review + + code-review: + type: agent + prompt: .ai/prompts/review.md + linear_state: active + runner: claude + model: opus + effort: xhigh + max_turns: 10 + session: fresh + transitions: + complete: delivery-review + + delivery-review: + type: gate + linear_state: review + rework_to: implement + max_rework: 5 + transitions: + approve: glean + + glean: + type: agent + prompt: .ai/prompts/glean.md + linear_state: active + runner: codex + model: gpt-5.6-sol + effort: xhigh + max_turns: 12 + session: fresh + transitions: + complete: glean-review + + glean-review: + type: gate + linear_state: review + rework_to: glean + max_rework: 3 + transitions: + approve: improvement + + # Mechanical: acts only on explicitly approved Glean candidate IDs. + improvement: + type: agent + prompt: .ai/prompts/improvement.md + linear_state: active + runner: codex + model: gpt-5.6-luna + effort: low + max_turns: 8 + session: fresh + transitions: + complete: merge + + # Mechanical: independently re-check approvals and CI, then merges. + merge: + type: agent + prompt: .ai/prompts/merge.md + linear_state: active + runner: claude + model: sonnet + effort: low + max_turns: 3 + session: fresh + transitions: + complete: done + + done: + type: terminal + linear_state: terminal diff --git a/.ai/workflows/spike.yaml b/.ai/workflows/spike.yaml new file mode 100644 index 0000000..6030044 --- /dev/null +++ b/.ai/workflows/spike.yaml @@ -0,0 +1,74 @@ +# ============================================================================= +# Spike — time-boxed technical investigation +# ============================================================================= + +description: "Technical uncertainty. The deliverable is a decision and approved next work." + +states: + investigate: + type: agent + prompt: .ai/prompts/explore.md + linear_state: active + runner: codex + model: gpt-5.6-sol + effort: xhigh + max_turns: 12 + session: inherit + transitions: + complete: ground-check + + ground-check: + type: agent + prompt: .ai/prompts/ground_check.md + linear_state: active + runner: codex + model: gpt-5.6-sol + effort: xhigh + max_turns: 8 + session: fresh + transitions: + complete: findings-review + + findings-review: + type: gate + linear_state: review + rework_to: investigate + max_rework: 3 + transitions: + approve: glean + + glean: + type: agent + prompt: .ai/prompts/glean.md + linear_state: active + runner: codex + model: gpt-5.6-sol + effort: xhigh + max_turns: 12 + session: fresh + transitions: + complete: glean-review + + glean-review: + type: gate + linear_state: review + rework_to: glean + max_rework: 3 + transitions: + approve: improvement + + improvement: + type: agent + prompt: .ai/prompts/improvement.md + linear_state: active + runner: codex + model: gpt-5.6-luna + effort: low + max_turns: 8 + session: fresh + transitions: + complete: done + + done: + type: terminal + linear_state: terminal diff --git a/workflow.glean.yaml b/workflow.glean.yaml new file mode 100644 index 0000000..d8665c8 --- /dev/null +++ b/workflow.glean.yaml @@ -0,0 +1,66 @@ +# Shared Stokowski runtime configuration. Pipelines live in workflows/*.yaml. + +tracker: + kind: linear + project_slug: "94a76f2ac65f" + assignee: me + +linear_states: + todo: "Todo" + active: "In Progress" + review: "Human Review" + gate_approved: "Gate Approved" + rework: "Rework" + terminal: [Done, Closed, Cancelled, Canceled, Duplicate] + +polling: + interval_ms: 15000 + +workspace: + root: ~/.local/share/stokowski/workspaces/linear-cli + +hooks: + after_create: | + git clone --depth 1 --recursive git@github.com:rubyists/linear-cli . + mise trust + mise install + mise exec -- mix setup + before_run: | + git fetch origin main + git rebase origin/main 2>/dev/null || git rebase --abort + timeout_ms: 240000 + +claude: + permission_mode: auto + max_turns: 20 + turn_timeout_ms: 3600000 + stall_timeout_ms: 300000 + +agent: + max_concurrent_agents: 4 + max_retry_backoff_ms: 300000 + max_concurrent_agents_by_state: + investigate: 2 + ground-check: 2 + implement: 2 + code-review: 1 + glean: 1 + improvement: 1 + merge: 1 + +prompts: + global_prompt: .ai/prompts/global.md + +server: + host: 127.0.0.1 + port: 4200 + +routing: + default: feature + rules: + - label: bug + workflow: bug-fix + - label: spike + workflow: spike + - label: exploration + workflow: exploration diff --git a/workflow.yaml b/workflow.yaml index 9791530..2db9b76 120000 --- a/workflow.yaml +++ b/workflow.yaml @@ -1 +1 @@ -workflow.opus.yaml \ No newline at end of file +workflow.glean.yaml \ No newline at end of file diff --git a/workflows b/workflows new file mode 120000 index 0000000..055fe73 --- /dev/null +++ b/workflows @@ -0,0 +1 @@ +.ai/workflows \ No newline at end of file