From c7182b80c104d50568c785f785a32d2c5fcbee3f Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 17:23:09 +0000 Subject: [PATCH 1/2] fix(docs-audit): a route anchor must name the route the page names MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The precision-first predicate listed `protocol/kernel/metadata-service.mdx` via `/environments/:environmentId`, a string that page does not contain. The row's attribution was correct; the anchor's own doc-side matcher is what widened. Two leniencies in `routePatternFor` were unconditional: - a parameter segment matched ANY `:name` spelling, so `:id` and `:environmentId` read as one anchor; - a concrete example value could fill a parameter with no static segment left to bound it, degrading the pattern to a prefix test. Both are narrowed. A parameter written as a parameter must name the same parameter (either spelling); a concrete value is admitted only where a static segment still follows in the tail. Requiring an end-of-path match was measured and rejected: it also deletes the four pages that document the route by its prefix. Separately, and as reporting only: a path or string literal that enters the diff inside a comment is still an anchor — excluding comments would drop the case where a JSDoc newly documenting a real route IS the signal — but its row now says the anchor came from a comment, so a reader can tell a prose mention from a registration without opening the page. Claude-Session: https://claude.ai/code/session_017ef78bLdybu3AffehKkhfk Co-authored-by: Claude --- scripts/docs-audit/affected-docs.mjs | 188 +++++++++++++++++++++++++-- 1 file changed, 176 insertions(+), 12 deletions(-) diff --git a/scripts/docs-audit/affected-docs.mjs b/scripts/docs-audit/affected-docs.mjs index 128841c2949..1ff4ef82d8b 100644 --- a/scripts/docs-audit/affected-docs.mjs +++ b/scripts/docs-audit/affected-docs.mjs @@ -257,11 +257,12 @@ const SELF_TEST_BATTERIES = Object.freeze({ '#11178: WHY a row is unreachable, and the two causes that printed as one': 28, '`causes` GETS THE SAME TREATMENT, AT BOTH ENDS (#11867)': 16, 'the ROUTE SOURCE concept: two kinds, and the runtime-registration guard (#11857)': 24, + 'ROUTE-ANCHOR PRECISION, IN BOTH DIRECTIONS (#16696)': 16, }); // DELETING an entry silences that battery's floor exactly as effectively as // zeroing it, so the roster's own size is pinned too. -const SELF_TEST_BATTERY_FLOOR = 29; +const SELF_TEST_BATTERY_FLOOR = 30; // The key an assertion is filed under when no battery is open. It is not a // declared battery, so it reds by the same set difference rather than silently @@ -1571,13 +1572,56 @@ function routeTailOf(literal) { * spellings a page may use — `:type`, `{type}`, or a concrete example value — so * `GET /api/v1/meta/object/account/history` in a code block still counts as documenting * `/:type/:name/history`. The static segments are what keep that from over-matching. + * + * THAT LAST SENTENCE IS A PRECONDITION, AND IT USED TO GO UNCHECKED (#16696). Both + * leniencies were unconditional, and each one on its own is enough to list a page for a + * route it does not document. Measured on PR #16694, where the anchor + * `/environments/:environmentId` listed `content/docs/protocol/kernel/metadata-service.mdx` + * — a page in which that string does not occur at all. The two hits that did it: + * + * :204 `/pub/v1/environments/:id/artifact[?commit=…]` — a DIFFERENT parameter name + * :213 `…/pub/v1/environments/env_42/artifact?commit=…` — a concrete value with the + * tail's own right edge open + * + * ⭐ This is MATCH WIDENING, not mis-attribution: the row named the anchor that really + * did select the page, and the anchor's own matcher is what accepted a path family the + * anchor never described. The fix is therefore on the matcher, and the two arms are + * narrowed separately because the two hits above fail for separate reasons: + * + * 1. A parameter WRITTEN AS A PARAMETER must name the SAME parameter. `:id` and + * `:environmentId` are two declared parameters, not two spellings of one, and a page + * that writes `/api/v1/cloud/environments/:id` is documenting the control-plane route + * it names, not the data-plane one this anchor came from. Either spelling of the + * right name still counts — `:environmentId` and `{environmentId}` are one parameter. + * 2. A CONCRETE EXAMPLE VALUE is admitted only where a STATIC segment still follows in + * the tail. That is the precondition above, made mechanical: a static segment to the + * right is what bounds the match, and a parameter with none is unbounded — filling it + * with `[A-Za-z0-9_%-]+` degrades the whole pattern to a prefix test on everything + * under `/environments/`, which is how `env_42` inside `/pub/v1/…/artifact` got in. + * `/:type/:name/history` keeps both concrete values, because `history` bounds them. + * + * ⛔ WHAT WAS MEASURED AND REJECTED, so nobody re-derives it: requiring the match to sit + * at the END of the documented path (the way the same tail is matched against a ledger + * row, `route.endsWith(tail)`). It kills both bad hits — and also kills + * `api/environment-routing.mdx`, the single most on-target page in that run, whose every + * occurrence is `/api/v1/environments/:environmentId/...` with a segment after it, plus + * `publish-and-preview.mdx`, `single-project-mode.mdx` and `http-protocol.mdx`. A route + * prefix written with its own parameter named IS a page documenting that route. */ function routePatternFor(tail) { const isParam = (s) => s.startsWith(':') || (s.startsWith('{') && s.endsWith('}')); - const body = tail - .split('/') - .filter(Boolean) - .map((s) => (isParam(s) ? '(?::[A-Za-z_$][\\w$]*|\\{[A-Za-z_$][\\w$]*\\}|[A-Za-z0-9_%-]+)' : s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'))) + const paramNameOf = (s) => (s.startsWith(':') ? s.slice(1) : s.slice(1, -1)); + const quote = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); + const segs = tail.split('/').filter(Boolean); + const body = segs + .map((s, i) => { + if (!isParam(s)) return quote(s); + const name = quote(paramNameOf(s)); + const named = `:${name}|\\{${name}\\}`; + // Arm 2: is anything static left to the RIGHT of this segment to bound a value? + const bounded = segs.slice(i + 1).some((rest) => !isParam(rest)); + return bounded ? `(?:${named}|[A-Za-z0-9_%-]+)` : `(?:${named})`; + }) .join('/'); return new RegExp(`/${body}(?![\\w-])`); } @@ -1652,7 +1696,29 @@ function isLiteralAnchorShape(lit) { || /^[A-Z][A-Z0-9]*(?:_[A-Z0-9]+)+$/.test(lit); // SCREAMING_SNAKE (#13471) } -/** Route tails and identifier-shaped string literals appearing on the changed lines. */ +/** + * Route tails and identifier-shaped string literals appearing on the changed lines. + * + * A COMMENT LINE IS A CHANGED LINE, AND THE ROW NOW SAYS SO (#16696). This scan reads the + * raw line on purpose — a path written in a doc comment is still evidence that the page + * documenting that path may need re-reading, and ⛔ excluding comments outright is the one + * shape ruled out: a JSDoc that newly documents a real route is sometimes exactly the + * signal. What it was NOT doing is telling the reader which kind of line it found. + * + * Measured on PR #16694: seven of that run's nine hand-written rows rode the anchor + * `/environments/:environmentId`, and that anchor entered the diff on exactly ONE added + * line — English prose inside a JSDoc block in `packages/client/src/index.ts`. The rows + * read `a path literal in meta`, indistinguishable from a route registration. A reader + * who opens seven pages and finds seven non-answers stops opening them, and the tool's + * whole value proposition is precision over the coarse package-mention fallback. + * + * So the STRENGTH of the hit rides in the provenance clause beside its location: an + * occurrence the comment mask blanks is reported as coming from a comment. Reporting only + * — ⛔ nothing here is admitted, dropped or reordered by it, so recall is unchanged by + * construction and a genuinely falsified page is still listed with a strong-hit anchor. + * A token found in code on one line and in a comment on another keeps BOTH clauses, the + * same way `noteFrom` already keeps every declaration a token reached through. + */ function literalAnchorsFromLines(lines, changed, surface = liveContainerSurface()) { const routes = new Set(); const literals = new Set(); @@ -1661,29 +1727,48 @@ function literalAnchorsFromLines(lines, changed, surface = liveContainerSurface( // mints none pays nothing. Reporting only: no route and no literal is admitted, // dropped or reordered by anything below. const from = new Map(); + // The code-only projection of the same lines, for the strength clause above. Blanking + // keeps every line and every offset (see `js-comment-mask`), so `masked[n - 1]` is the + // same line with its prose removed and nothing else moved. + const masked = maskComments(lines.join('\n')).split('\n'); + const ROUTE_RE = /(?:\/[A-Za-z0-9_:.$*{}-]+){2,}/g; + const LITERAL_RE = /['"]([A-Za-z][\w.$-]{3,63})['"]/g; for (const n of changed) { const line = lines[n - 1]; if (line === undefined) continue; + const code = masked[n - 1] ?? line; let enclosing; const enclosingName = () => { if (enclosing === undefined) enclosing = documentableDeclarationsAt(lines, n - 1, surface)[0] || null; return enclosing ? enclosing.name : null; }; - for (const m of line.replace(/\$\{[^}]*\}/g, '').matchAll(/(?:\/[A-Za-z0-9_:.$*{}-]+){2,}/g)) { + // What the SAME two scans find once the prose is blanked. Set membership, not offsets: + // the question a row answers is "did this anchor have a code occurrence on this line", + // and a tail present in both projections is code wherever else it also appears. + const codeRoutes = new Set(); + for (const m of code.replace(/\$\{[^}]*\}/g, '').matchAll(ROUTE_RE)) { + const t = routeTailOf(m[0]); + if (t) codeRoutes.add(t); + } + const codeLiterals = new Set(); + for (const m of code.matchAll(LITERAL_RE)) codeLiterals.add(m[1]); + const site = (kind, where, inComment) => { + const lead = inComment ? `a ${kind} in a comment` : `a ${kind}`; + return where ? `${lead} in ${where}` : `${lead} on a changed line`; + }; + for (const m of line.replace(/\$\{[^}]*\}/g, '').matchAll(ROUTE_RE)) { const tail = routeTailOf(m[0]); if (tail) { routes.add(tail); - const where = enclosingName(); - noteFrom(from, tail, where ? `a path literal in ${where}` : 'a path literal on a changed line'); + noteFrom(from, tail, site('path literal', enclosingName(), !codeRoutes.has(tail))); } } - for (const m of line.matchAll(/['"]([A-Za-z][\w.$-]{3,63})['"]/g)) { + for (const m of line.matchAll(LITERAL_RE)) { const lit = m[1]; if (GENERIC_ANCHOR_NAMES.has(lit.toLowerCase())) continue; if (!isLiteralAnchorShape(lit)) continue; literals.add(lit); - const where = enclosingName(); - noteFrom(from, lit, where ? `a string literal in ${where}` : 'a string literal on a changed line'); + noteFrom(from, lit, site('string literal', enclosingName(), !codeLiterals.has(lit))); } } return { routes, literals, from }; @@ -5739,6 +5824,85 @@ function selfTest() { 'git ls-files packages, filtered to .ts under dot-directories', 0, trackedDotDirFiles.length); } + battery('ROUTE-ANCHOR PRECISION, IN BOTH DIRECTIONS (#16696)'); + // ⛔ THIS BATTERY IS TWO-DIRECTIONAL BY CONSTRUCTION, and the second direction is the + // load-bearing one. A change that made the bad row disappear by matching less, or by + // dropping comment-sourced anchors, would turn this card's findings green while + // deleting the thing the tool is for — a precision-first list whose rows a reader + // trusts enough to act on. So every narrowing case below is paired with a KEEP case + // taken from the same run, and the provenance cases assert that a clause was ADDED, + // never that an anchor was removed. + + // ── Direction 1: the wrong row can no longer be produced ────────────────── + // Both strings are verbatim from `content/docs/protocol/kernel/metadata-service.mdx` + // (:204 and :213 on `8cf527f8e`), the page PR #16694 listed via an anchor it does not + // contain. The instrument's own positive control is the row below them: the same + // matcher, the same page's path family, spelled with the parameter the anchor names. + const routePrecisionCases = [ + ['/environments/:environmentId', 'artifact route (`/pub/v1/environments/:id/artifact[?commit=]`) serves', false, + 'a DIFFERENT parameter name is a different route — #16696 finding 1, hit A'], + ['/environments/:environmentId', "path: 'https://cloud.example.com/pub/v1/environments/env_42/artifact?commit=cmt_1a2b',", false, + 'a concrete value cannot fill an UNBOUNDED trailing parameter — #16696 finding 1, hit B'], + ['/environments/:environmentId', 'Route REST, metadata, automation, AI, and package calls through /api/v1/environments/:environmentId/....', true, + 'POSITIVE CONTROL: the page that DOES name it stays listed (api/environment-routing.mdx:3)'], + ['/environments/:environmentId', '| `auto` | Registers both unscoped `/api/v1/...` and scoped `/api/v1/environments/:environmentId/...` routes. |', true, + 'a route PREFIX with its own parameter named is still a page documenting that route'], + ['/environments/:environmentId', '`/api/v1/environments/{environmentId}`', true, + 'the brace spelling of the SAME name is one parameter, not two'], + ['/environments/:environmentId', 'Cloud control-plane endpoints such as `/api/v1/cloud/environments/:id` manage', false, + 'a complete path with a different parameter name is still a different route'], + // The designed leniency, untouched: `history` bounds both parameters on the right. + ['/:type/:name/history', 'GET /api/v1/meta/object/account/history', true, + 'concrete values stay admitted where a STATIC segment bounds them'], + ['/:type/:name/history', 'GET /api/v1/meta/{type}/{name}/history', true, + 'the brace spelling stays admitted'], + ['/:type/:name/history', 'GET /api/v1/meta/object/account/audit', false, + 'a different static segment still does not match'], + ]; + for (const [tail, text, want, label] of routePrecisionCases) { + check('routePatternFor', label, `${tail} vs ${JSON.stringify(text)}`, want, routePatternFor(tail).test(text)); + } + + // ── Direction 2: strength is REPORTED, and nothing is dropped for it ────── + // The fixture is the shape PR #16694 actually carried: a path that enters the diff + // only as English prose in a JSDoc block, and — on a separate line — a path that + // enters as a real registration. Both must anchor; only the first may say "comment". + const provenanceSource = [ + 'const routes = {', + ' /**', + ' * Reaches the same handler as the unscoped twin — one replay against', + ' * `/environments/:environmentId` — so the body is byte-identical.', + ' */', + " register: (app) => app.get('/api/v1/meta/:type/:name/history', historyHandler),", + '};', + ]; + const prov = literalAnchorsFromLines(provenanceSource, [1, 2, 3, 4, 5, 6, 7]); + const clausesFor = (tail) => [...(prov.from.get(tail) || [])].join(' | '); + check('literalAnchorsFromLines', 'the comment-only path is STILL an anchor — ⛔ not excluded (#16696 finding 2)', + '/environments/:environmentId', true, prov.routes.has('/environments/:environmentId')); + check('literalAnchorsFromLines', '…and its row says the anchor came from a comment', + 'clause for /environments/:environmentId', true, /in a comment/.test(clausesFor('/environments/:environmentId'))); + check('literalAnchorsFromLines', 'the registered path is an anchor too', + '/api/v1/meta/:type/:name/history', true, prov.routes.has('/api/v1/meta/:type/:name/history')); + check('literalAnchorsFromLines', '…and a STRONG hit says nothing about comments — the reader can tell them apart', + 'clause for /api/v1/meta/:type/:name/history', false, /in a comment/.test(clausesFor('/api/v1/meta/:type/:name/history'))); + // A token in code on one line and in prose on another keeps BOTH clauses: `noteFrom` + // accumulates, and a row that printed only one of them would read like the only answer. + const bothSource = [ + " // documented at `/api/v1/meta/:type/:name/history`", + " app.get('/api/v1/meta/:type/:name/history', historyHandler);", + ]; + const both = [...(literalAnchorsFromLines(bothSource, [1, 2]).from.get('/api/v1/meta/:type/:name/history') || [])]; + check('literalAnchorsFromLines', 'code and comment on two lines keep BOTH clauses', + 'clause count for a tail written in each', 2, both.length); + check('literalAnchorsFromLines', '…one of which is the comment clause', + 'at least one clause names a comment', 1, both.filter((c) => /in a comment/.test(c)).length); + // A string literal inside a comment rides the same predicate — same loop, same mask. + const litComment = literalAnchorsFromLines([" // the stored key is 'controlled_by_parent' on the overlay"], [1]); + check('literalAnchorsFromLines', 'a string literal in prose is kept AND marked', + 'controlled_by_parent', true, litComment.literals.has('controlled_by_parent') + && /in a comment/.test([...(litComment.from.get('controlled_by_parent') || [])].join(' | '))); + // ── The floor: every declared battery RAN, and ran its cases (#13489) ─── // // Evaluated after every battery has had its chance and BEFORE the verdict, so From fef76a4aaf1cccbead6896e208db8d29b5f8fd13 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 17:36:21 +0000 Subject: [PATCH 2/2] fix(docs-audit): scope the route-matcher narrowing to the two hits that were wrong MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The first cut refused a concrete example value in ANY parameter with no static segment behind it in the tail. Measured over the 228 declared route tails against all 195 hand-written docs, that reads -36.3% and takes the good rows with it: `/data/:object` stops matching a page that writes `POST /api/v1/data/accounts`, which is the leniency working as designed. Only the tail's LAST segment has nothing behind it in the pattern, so only there does a concrete value have to end the documented path. Re-measured: 571 -> 524 rows, -8.2%, zero tails going from matching some page to matching none, and the 42 rows arm 2 drops are one shape — a page documenting a LONGER route, or a monorepo source path (`packages/core`) matched as the wire route `/api/v1/packages/:id`. README carries both narrowings, the rejected variant with its number, and the comment-provenance clause. Claude-Session: https://claude.ai/code/session_017ef78bLdybu3AffehKkhfk Co-authored-by: Claude --- scripts/docs-audit/README.md | 63 +++++++++++++++++++++++++++- scripts/docs-audit/affected-docs.mjs | 59 ++++++++++++++++++-------- 2 files changed, 102 insertions(+), 20 deletions(-) diff --git a/scripts/docs-audit/README.md b/scripts/docs-audit/README.md index 20f219eec5c..08be744a929 100644 --- a/scripts/docs-audit/README.md +++ b/scripts/docs-audit/README.md @@ -87,9 +87,9 @@ Each anchor kind names its own origin, from the same field the JSON publishes as | kind | the clause | |:--|:--| | `symbol` | `a field of interface MetaOverlayCacheKey` · `a method of class RestServer` · `a top-level function` | -| `route` | `a path literal in RestServer` · `bridged from symbol enforceEnvironmentOwnership — its route-source handler names it` | +| `route` | `a path literal in RestServer` · `a path literal in a comment in meta` · `bridged from symbol enforceEnvironmentOwnership — its route-source handler names it` | | `sdk` | `the route ledger binds it to GET /api/v1/ui/view/:object/:type` | -| `literal` | `a string literal in cacheKeyOf` | +| `literal` | `a string literal in cacheKeyOf` · `a string literal in a comment in cacheKeyOf` | | `command` | `read off packages/cli/src/commands/environments/bind.ts` | | `rule` | `a @docs-rule block in packages/objectql/src/engine.ts` | @@ -108,6 +108,65 @@ cannot start deciding with it without going red. Container-qualified *discrimina the separate, later step below (option D), and it decides from the declarations directly, never by parsing this clause back out. +### A row says whether its anchor came from CODE or from PROSE (#16696) + +A comment line is a changed line, so a path written in a JSDoc mints a `route` anchor like +any other. That is deliberate and stays: a doc comment that newly documents a real route is +sometimes exactly the signal that the page documenting it needs re-reading, and ⛔ excluding +comments from anchor sources is the one repair ruled out. What was missing is the reader's +half — the row did not say which kind of line it found. + +Measured on PR #16694: seven of that run's nine hand-written rows rode +`/environments/:environmentId`, and that anchor entered the diff on **exactly one added +line**, English prose inside a JSDoc block in `packages/client/src/index.ts`. Every row read +`a path literal in meta`, indistinguishable from a registration. A reader who opens seven +pages and finds seven non-answers stops opening them, and then the list fails on the PR +where it is right. + +So the clause carries the strength beside the location: `a path literal in a comment in meta` +against `a path literal in meta`. Like every other clause it is **publication, not +discrimination** — the mask is read only to word the clause, nothing is admitted or dropped +by it, and `--self-test` pins both directions: the comment-only anchor is still an anchor, +and a code occurrence still produces a clause that says nothing about comments. + +### A route anchor must name the route the page names (#16696) + +The doc-side matcher lets a page spell a parameter three ways — `:type`, `{type}`, or a +concrete example value — so `GET /api/v1/meta/object/account/history` counts as documenting +`/:type/:name/history`. Both leniencies used to be unconditional, and each one alone is +enough to list a page for a route it does not document. `protocol/kernel/metadata-service.mdx` +was listed via `/environments/:environmentId`, a string that page does not contain, on: + +``` +:204 `/pub/v1/environments/:id/artifact[?commit=]` — a DIFFERENT parameter name +:213 'https://cloud.example.com/pub/v1/environments/env_42/artifact?commit=cmt_1a2b' + — a concrete value with the tail's right edge open +``` + +Two narrowings, one per hit. A parameter **written as a parameter** must name the same +parameter, either spelling. A **concrete value filling the tail's last segment** must end the +documented path — every earlier parameter is still bounded by the segments to its right, so +`GET /api/v1/data/accounts` still matches `/data/:object` while +`GET /api/v1/data/account/123` no longer does, because that is `/data/:object/:id` and that +is the tail which lists it. + +Measured over the **228** distinct route tails declared in this repo's **28** route-ledger +files against all **195** hand-written docs: tail×page rows **571 → 524**, −8.2%, with **zero** +tails going from matching some page to matching none. Arm 1 accounts for 5 of the 47 dropped +rows; the other 42 are one shape — `/packages/:id` matched eleven pages on the monorepo +source paths `packages/core`, `packages/spec`, `packages/plugins` …, and `/meta/:type` +matched `api/environment-routing.mdx` on the prose *"data/meta/AI/automation"*. + +⛔ Requiring **every** match to sit at the end of the documented path — the way the same tail +is matched against a ledger row — was measured and rejected: −36.3%, and it deletes +`api/environment-routing.mdx`, the most on-target page of that run, whose every occurrence is +`/api/v1/environments/:environmentId/...` with a segment after it. A route PREFIX written +with the route's own parameter named IS a page documenting that route. + +⛔ **No claim is made here about how often either shape occurs.** One PR is nine rows, not a +population; the corpus figures above are about the matcher over the declared route surface, +which is a different statement and is the only one measured. + ### A data property is qualified by its declaring container (#13713) Option D of the same ruling, and the half that *does* move rows. It landed once the spec diff --git a/scripts/docs-audit/affected-docs.mjs b/scripts/docs-audit/affected-docs.mjs index 1ff4ef82d8b..6010e3c025e 100644 --- a/scripts/docs-audit/affected-docs.mjs +++ b/scripts/docs-audit/affected-docs.mjs @@ -257,7 +257,7 @@ const SELF_TEST_BATTERIES = Object.freeze({ '#11178: WHY a row is unreachable, and the two causes that printed as one': 28, '`causes` GETS THE SAME TREATMENT, AT BOTH ENDS (#11867)': 16, 'the ROUTE SOURCE concept: two kinds, and the runtime-registration guard (#11857)': 24, - 'ROUTE-ANCHOR PRECISION, IN BOTH DIRECTIONS (#16696)': 16, + 'ROUTE-ANCHOR PRECISION, IN BOTH DIRECTIONS (#16696)': 20, }); // DELETING an entry silences that battery's floor exactly as effectively as @@ -1593,20 +1593,34 @@ function routeTailOf(literal) { * that writes `/api/v1/cloud/environments/:id` is documenting the control-plane route * it names, not the data-plane one this anchor came from. Either spelling of the * right name still counts — `:environmentId` and `{environmentId}` are one parameter. - * 2. A CONCRETE EXAMPLE VALUE is admitted only where a STATIC segment still follows in - * the tail. That is the precondition above, made mechanical: a static segment to the - * right is what bounds the match, and a parameter with none is unbounded — filling it - * with `[A-Za-z0-9_%-]+` degrades the whole pattern to a prefix test on everything - * under `/environments/`, which is how `env_42` inside `/pub/v1/…/artifact` got in. - * `/:type/:name/history` keeps both concrete values, because `history` bounds them. - * - * ⛔ WHAT WAS MEASURED AND REJECTED, so nobody re-derives it: requiring the match to sit - * at the END of the documented path (the way the same tail is matched against a ledger - * row, `route.endsWith(tail)`). It kills both bad hits — and also kills - * `api/environment-routing.mdx`, the single most on-target page in that run, whose every - * occurrence is `/api/v1/environments/:environmentId/...` with a segment after it, plus + * 2. A CONCRETE EXAMPLE VALUE filling the tail's LAST segment must END the documented + * path. Every earlier parameter is bounded by the rest of the pattern — the segments + * to its right still have to be there — but the last one has nothing behind it, so a + * value there degrades the whole pattern to a prefix test: `/environments/`, + * which is how `env_42` inside `/pub/v1/…/artifact` got in. A page writing + * `GET /api/v1/data/accounts` still matches `/data/:object`, because the path ends + * where the anchor does; a page writing `GET /api/v1/data/account/123` no longer + * does, because it is documenting `/data/:object/:id` — a different route, and the + * one that lists it when the diff touches it. + * + * MEASURED, on `c7182b80c`, over the 228 distinct route tails declared in this repo's 28 + * route-ledger files against all 195 hand-written docs (the denominator is + * `handwritten-docs.json`, the same one the tool's recall figures use): tail×page rows + * 571 → 524, −8.2%, with ZERO tails going from matching some page to matching none. Arm 1 + * accounts for 5 of the 47 dropped rows and arm 2 for 42, and the 42 are one shape — + * a page documenting a LONGER route than the anchor, or not a route at all: + * `/packages/:id` matched eleven pages on the monorepo source paths `packages/core`, + * `packages/spec`, `packages/plugins` …, and `/meta/:type` matched + * `api/environment-routing.mdx` on the prose "data/meta/AI/automation". + * + * ⛔ WHAT WAS MEASURED AND REJECTED, so nobody re-derives it: requiring EVERY match to sit + * at the end of the documented path (the way the same tail is matched against a ledger + * row, `route.endsWith(tail)`). That reads −36.3% and it kills `api/environment-routing.mdx`, + * the single most on-target page in the #16694 run, whose every occurrence is + * `/api/v1/environments/:environmentId/...` with a segment after it — plus * `publish-and-preview.mdx`, `single-project-mode.mdx` and `http-protocol.mdx`. A route - * prefix written with its own parameter named IS a page documenting that route. + * PREFIX written with the route's own parameter named IS a page documenting that route; + * that is why arm 2 is scoped to a concrete VALUE in the LAST position and nothing else. */ function routePatternFor(tail) { const isParam = (s) => s.startsWith(':') || (s.startsWith('{') && s.endsWith('}')); @@ -1618,9 +1632,10 @@ function routePatternFor(tail) { if (!isParam(s)) return quote(s); const name = quote(paramNameOf(s)); const named = `:${name}|\\{${name}\\}`; - // Arm 2: is anything static left to the RIGHT of this segment to bound a value? - const bounded = segs.slice(i + 1).some((rest) => !isParam(rest)); - return bounded ? `(?:${named}|[A-Za-z0-9_%-]+)` : `(?:${named})`; + // Arm 2: only the LAST segment has nothing behind it in the pattern to bound a + // concrete value, so only there does the documented path have to end. + const value = i === segs.length - 1 ? '[A-Za-z0-9_%-]+(?![\\w/-])' : '[A-Za-z0-9_%-]+'; + return `(?:${named}|${value})`; }) .join('/'); return new RegExp(`/${body}(?![\\w-])`); @@ -5842,7 +5857,15 @@ function selfTest() { ['/environments/:environmentId', 'artifact route (`/pub/v1/environments/:id/artifact[?commit=]`) serves', false, 'a DIFFERENT parameter name is a different route — #16696 finding 1, hit A'], ['/environments/:environmentId', "path: 'https://cloud.example.com/pub/v1/environments/env_42/artifact?commit=cmt_1a2b',", false, - 'a concrete value cannot fill an UNBOUNDED trailing parameter — #16696 finding 1, hit B'], + 'a concrete value in the LAST segment must end the documented path — #16696 finding 1, hit B'], + ['/data/:object', 'Create a record with `POST /api/v1/data/accounts` and read it back.', true, + 'POSITIVE CONTROL for arm 2: a concrete value that DOES end the path still matches'], + ['/data/:object', 'Incoming API request: GET /api/v1/data/account/123', false, + 'the same page text one segment longer is documenting /data/:object/:id, not this route'], + ['/data/:object/:id', 'Incoming API request: GET /api/v1/data/account/123', true, + '…and THAT tail is the one that lists it — the row moves, the page is not lost'], + ['/packages/:id', 'see [`packages/core`](https://github.com/o/r/blob/main/packages/core/src/x.ts)', false, + 'a monorepo source path is not the wire route /api/v1/packages/:id'], ['/environments/:environmentId', 'Route REST, metadata, automation, AI, and package calls through /api/v1/environments/:environmentId/....', true, 'POSITIVE CONTROL: the page that DOES name it stays listed (api/environment-routing.mdx:3)'], ['/environments/:environmentId', '| `auto` | Registers both unscoped `/api/v1/...` and scoped `/api/v1/environments/:environmentId/...` routes. |', true,