From 6d2602ff26880ac02bb2f736617ef5acb09a8932 Mon Sep 17 00:00:00 2001 From: Jiarui Xu <39042389+jxudata@users.noreply.github.com> Date: Mon, 5 Oct 2026 09:08:34 -0700 Subject: [PATCH 1/3] Fix explainer routing and clarify single-pass inference --- materials/tabicl-explainer/README.md | 18 +++--- .../src/components/AttentionMatrix.svelte | 6 +- .../src/components/Sankey.svelte | 20 ++++--- .../src/components/textbook/Textbook.svelte | 12 ++-- materials/tabicl-explainer/src/lib/tabicl.ts | 4 +- .../tabicl-explainer/src/routes/+page.svelte | 57 ++++++++++--------- materials/website/index.html | 29 +++++----- materials/website/js/site.js | 16 +++++- materials/website/js/tabicl/nanotabicl.js | 2 +- materials/website/model/PROVENANCE.md | 14 +++-- tests/tabicl-browser-runtime.test.mjs | 12 ++++ 11 files changed, 117 insertions(+), 73 deletions(-) diff --git a/materials/tabicl-explainer/README.md b/materials/tabicl-explainer/README.md index ae022cf..93af9f6 100644 --- a/materials/tabicl-explainer/README.md +++ b/materials/tabicl-explainer/README.md @@ -3,9 +3,11 @@ Interactive visualization of TabICLv2 inference on fixed UCI Iris examples. The 12-row context is intentionally compact for tracing. It is outside the -officially documented TabICLv2 pretraining range of 300 to 48K rows, and the -upstream authors state that sub-300-row generalization has not been tested. -Treat its measured output as an out-of-regime illustration, not a quality claim. +officially documented TabICLv2 pretraining range of 300 to 48K rows. The revised +paper includes sub-300-row few-shot evaluations +([Appendix L.3–L.4, September 2026 revision](https://arxiv.org/html/2602.11139v2#A12.SS3)). +Those results do not establish the quality of this particular 12-row Iris demo. +Treat its measured output as an illustration of computation, not a quality claim. The interface adapts the MIT-licensed [Transformer Explainer](https://github.com/poloclub/transformer-explainer) layout while replacing GPT-2 generation with tabular in-context learning. @@ -55,10 +57,12 @@ The original interface is used under the MIT License reproduced in This adaptation preserves the interface composition while replacing GPT inference and examples with fixed UCI Iris records and a browser TabICLv2 -checkpoint. The explorer uses a nanoTabICL-derived bridge for selected-view -inspection with the released TabICLv2 checkpoint's feature-group offsets. It is -not the standalone nanoTabICL model and does not expose every preprocessing and -ensemble option in the official `TabICLClassifier`. +checkpoint. The explorer uses a nanoTabICL-derived bridge for fixed single-pass +core inspection with the released TabICLv2 checkpoint's feature-group offsets. It is +not the standalone nanoTabICL model. It uses context-only z-score standardization, +original feature and class order, and softmax temperature 1. It is not a selected +view of the main playground's eight-view `TabICLClassifier`, which applies its own +preprocessing, permutations, and temperature 0.9. Checkpoint source, hashes, runtime behavior, and limitations are documented in [`../website/model/PROVENANCE.md`](../website/model/PROVENANCE.md). TabICL code diff --git a/materials/tabicl-explainer/src/components/AttentionMatrix.svelte b/materials/tabicl-explainer/src/components/AttentionMatrix.svelte index 62d28f8..2f3aef0 100644 --- a/materials/tabicl-explainer/src/components/AttentionMatrix.svelte +++ b/materials/tabicl-explainer/src/components/AttentionMatrix.svelte @@ -27,8 +27,7 @@ return (value: number) => (Number.isFinite(value) ? scale(value) : '#f3f4f6'); }; - $: signedColor = createSignedColor(raw ?? []); - $: scaledColor = createSignedColor(scaled ?? []); + $: signedColor = createSignedColor([...(raw ?? []), ...(scaled ?? [])]); const weightColor = (value: number) => Number.isFinite(value) ? d3.interpolateRgb('#ffffff', '#6d28d9')(Math.min(1, value * 4)) : '#f3f4f6'; @@ -39,6 +38,7 @@ class:expanded on:click={() => (expanded = !expanded)} aria-expanded={expanded} + aria-label={expanded ? 'Collapse attention calculation; logits share a color scale' : 'Expand attention calculation'} > {#if expanded}
@@ -48,7 +48,7 @@
→
- +
QASSMax-scaled logits
QASSMax(Q) · Kᵀ / √d
diff --git a/materials/tabicl-explainer/src/components/Sankey.svelte b/materials/tabicl-explainer/src/components/Sankey.svelte index 40184dd..2a15be9 100644 --- a/materials/tabicl-explainer/src/components/Sankey.svelte +++ b/materials/tabicl-explainer/src/components/Sankey.svelte @@ -9,16 +9,20 @@ const draw = async () => { await tick(); if (!svgEl) return; - const host = svgEl.parentElement?.getBoundingClientRect(); - const nodes = Array.from(document.querySelectorAll('[data-flow-node]')); - if (!host || nodes.length < 2) return; + const host = svgEl.parentElement; + const matrix = svgEl.getScreenCTM(); + const nodes = Array.from(host?.querySelectorAll('[data-flow-node]') ?? []); + if (!matrix || nodes.length < 2) return; + // Bounding rectangles use screen coordinates, including the app's CSS scale. + // Convert back to SVG coordinates so the scale is applied only once. + const inverse = matrix.inverse(); const links = nodes.slice(0, -1).map((node, index) => { const source = node.getBoundingClientRect(); const target = nodes[index + 1].getBoundingClientRect(); - const x1 = source.right - host.left; - const x2 = target.left - host.left; - const y1 = source.top + source.height / 2 - host.top; - const y2 = target.top + target.height / 2 - host.top; + const start = new DOMPoint(source.right, source.top + source.height / 2).matrixTransform(inverse); + const end = new DOMPoint(target.left, target.top + target.height / 2).matrixTransform(inverse); + const { x: x1, y: y1 } = start; + const { x: x2, y: y2 } = end; const curve = Math.max(20, (x2 - x1) * 0.45); return { path: `M${x1},${y1} C${x1 + curve},${y1} ${x2 - curve},${y2} ${x2},${y2}`, @@ -38,10 +42,12 @@ observer = new ResizeObserver(draw); const host = svgEl.parentElement; if (host) observer.observe(host); + host?.addEventListener('transitionend', draw); window.addEventListener('resize', draw); draw(); return () => { observer.disconnect(); + host?.removeEventListener('transitionend', draw); window.removeEventListener('resize', draw); }; }); diff --git a/materials/tabicl-explainer/src/components/textbook/Textbook.svelte b/materials/tabicl-explainer/src/components/textbook/Textbook.svelte index ef69322..dc14d4a 100644 --- a/materials/tabicl-explainer/src/components/textbook/Textbook.svelte +++ b/materials/tabicl-explainer/src/components/textbook/Textbook.svelte @@ -5,7 +5,7 @@ const pages = [ { title: 'A table becomes a prediction', - body: 'This nanoTabICL-shaped inspector receives twelve labeled Iris examples and one unlabeled query using mapped official TabICLv2 weights. It is not the standalone nanoTabICL model. This trace is outside the documented 300 to 48K-row pretraining regime and is not evidence of expected model quality. Choose Query A, B, or C above. Its hidden species is never supplied to the model.' + body: 'This inspector runs a fixed single core pass with quantized official TabICLv2 weights on twelve labeled Iris examples and one unlabeled query. The context is below the documented 300 to 48K-row pretraining range. Choose Query A, B, or C; its label is never supplied. This illustrates computation, not model quality.' }, { title: '1 · Read the table', @@ -13,23 +13,23 @@ }, { title: '2 · Standardize, group, embed', - body: 'Feature values are standardized with context-only mean and standard deviation, circularly grouped at offsets 1, 2, and 4, then projected into 128-dimensional column tokens.' + body: 'This core pass uses context-only mean and standard deviation, feature groups at checkpoint offsets 1, 2, and 4, and a projection to 128 dimensions per group. Context tokens also receive label embeddings; the query does not. Activation colors are scaled separately for each tensor and cannot be compared across panels.' }, { title: '3 · Column attention', - body: 'Three induced Column Transformer blocks process each feature column. The colored strips are sampled live activations; the thin repeated cards describe symbolic architecture.' + body: 'Context rows first build three blocks of inducing summaries. The query reads cached summaries through the second attention operation of each block; it does not rebuild them. Colored strips show sampled query activations. The inducing cards are symbolic.' }, { title: '4 · Compress each row', - body: 'Three Row Transformer blocks combine feature tokens with four learned CLS tokens. The final four CLS outputs concatenate into one 512-dimensional row vector.' + body: 'Three Row Transformer blocks combine grouped-feature tokens with four learned CLS tokens. The final CLS outputs are normalized and concatenated into a 512-dimensional row vector. Context vectors receive a second label embedding before dataset-level ICL; the query does not.' }, { title: '5 · Route through context', - body: 'Twelve ICL Transformer blocks let Q attend to E1–E12. Attention is routing, not feature importance. Change block and head, or expand the matrix to inspect real logits and weights.' + body: 'Twelve ICL Transformer blocks let Q attend to cached context keys and values from E1–E12. Attention weights describe routing, not causal attribution. Change block or head, or expand to compare logits on a shared color scale and inspect softmax weights on their separate scale.' }, { title: '6 · Predict the class', - body: 'The final query vector passes through output normalization and an MLP. The three bars are computed softmax probabilities for one selected TabICLv2 model view using mapped official weights.' + body: 'Output normalization and an MLP produce three class logits, converted to probabilities with temperature 1. This fixed core pass keeps the original feature and class order. It does not select one of the playground classifier’s eight views or use its temperature 0.9.' } ]; diff --git a/materials/tabicl-explainer/src/lib/tabicl.ts b/materials/tabicl-explainer/src/lib/tabicl.ts index 2e14be6..108ee59 100644 --- a/materials/tabicl-explainer/src/lib/tabicl.ts +++ b/materials/tabicl-explainer/src/lib/tabicl.ts @@ -125,7 +125,7 @@ async function fetchWithProgress( export async function loadTabICL( onStatus: (status: LoadStatus) => void ): Promise<{ model: TabICLModel; cache: unknown }> { - onStatus({ phase: 'downloading', progress: 0, message: 'Downloading official TabICLv2 weights' }); + onStatus({ phase: 'downloading', progress: 0, message: 'Downloading quantized TabICLv2 weights' }); const modelBase = new URL('../model/', window.location.href); const manifestResponse = await fetch(new URL('manifest.json', modelBase), { cache: 'no-store' }); if (!manifestResponse.ok) throw new Error(`Manifest download failed (${manifestResponse.status})`); @@ -150,6 +150,6 @@ export async function loadTabICL( CONTEXT.map((row) => [...row.x]), CONTEXT.map((row) => row.y) ); - onStatus({ phase: 'ready', progress: 1, message: 'nanoTabICL trace ready' }); + onStatus({ phase: 'ready', progress: 1, message: 'TabICLv2 core trace ready' }); return { model, cache }; } diff --git a/materials/tabicl-explainer/src/routes/+page.svelte b/materials/tabicl-explainer/src/routes/+page.svelte index 0ea8d9b..3c48348 100644 --- a/materials/tabicl-explainer/src/routes/+page.svelte +++ b/materials/tabicl-explainer/src/routes/+page.svelte @@ -36,22 +36,25 @@ const DESIGN_WIDTH = 1880; let fitScale = 1; let designHeight = 900; + let inspectionRequest = 0; $: query = QUERIES.find((item) => item.id === selectedQuery) ?? QUERIES[0]; $: selectedAttention = inspection?.iclBlocks[iclBlock]?.attention ?? null; $: selectedColumn = inspection?.columnBlocks[columnBlock]?.activation ?? null; $: selectedRow = inspection?.rowBlocks[rowBlock]?.tokens ?? null; - $: revision = `${selectedQuery}-${columnBlock}-${rowBlock}-${iclBlock}-${attentionHead}-${loadStatus.phase}`; + $: revision = `${selectedQuery}-${columnBlock}-${rowBlock}-${iclBlock}-${attentionHead}-${loadStatus.phase}-${guidePage}-${textbookOpen}-${fitScale}`; async function runInspection(activeModel: TabICLModel, activeCache: unknown, queryId: string) { + const request = ++inspectionRequest; const selected = QUERIES.find((item) => item.id === queryId) ?? QUERIES[0]; inspection = null; loadStatus = { phase: 'running', progress: 0.98, message: `Running real Query ${queryId}` }; await tick(); await new Promise((resolve) => setTimeout(resolve, 20)); + if (request !== inspectionRequest) return; try { inspection = activeModel.inspectQuery(activeCache, [...selected.x], 3); - loadStatus = { phase: 'ready', progress: 1, message: 'nanoTabICL trace ready' }; + loadStatus = { phase: 'ready', progress: 1, message: 'TabICLv2 core trace ready' }; } catch (error) { loadStatus = { phase: 'error', @@ -99,9 +102,9 @@ return `rgb(${target.map((channel) => Math.round(255 + (channel - 255) * amount)).join(',')})`; } - function stageClass(page: number) { - if (!textbookOpen || guidePage === 0) return ''; - return guidePage === page ? 'spotlit' : 'muted'; + function stageClass(page: number, activePage: number, open: boolean) { + if (!open || activePage === 0) return ''; + return activePage === page ? 'spotlit' : 'muted'; } function cycle(value: number, delta: number, length: number) { @@ -113,7 +116,7 @@ TabICL Explainer: Tabular In-Context Learning, Visually Explained @@ -128,7 +131,7 @@
-
+
IRIS TABLE
real UCI records · cm
@@ -153,7 +156,7 @@
Query {selectedQuery} is visibly unlabeled
-
+
PREPROCESS + EMBED
live query path
@@ -172,13 +175,13 @@
Wx Linear embed - 4 × 128 + output: 4 groups × 128
LIVE TENSOR SAMPLE
{#each inspection?.embedding.values ?? Array(4).fill(Array(24).fill(Number.NaN)) as values, index}
- f{index + 1} + g{index + 1}
{#each values as value}
-
+
COLUMN TRANSFORMER
induced attention · 8 heads
@@ -202,9 +205,9 @@ {/each}
- + Block {columnBlock + 1} of 3 - +
128 inducing vectors
TFM 1QASSMax
@@ -213,7 +216,7 @@
{#each selectedColumn?.values ?? Array(4).fill(Array(24).fill(Number.NaN)) as values, index}
- f{index + 1} + g{index + 1}
{#each values as value}
-
+
ROW TRANSFORMER
feature mixing + CLS compression
- + Block {rowBlock + 1} of 3 - +
{#each Array(4) as _, index}
CLS {index + 1}
{/each} - {#each Array(rowBlock === 2 ? 0 : 4) as _, index}
f{index + 1}
{/each} + {#each Array(rowBlock === 2 ? 0 : 4) as _, index}
g{index + 1}
{/each}
8-head self-attention + MLP
@@ -256,7 +259,7 @@
-
+
ROW VECTORS
4 CLS × 128 → 512
@@ -280,7 +283,7 @@
-
+
ICL TRANSFORMER × 12
query-to-context routing
@@ -290,14 +293,14 @@
- + Block {iclBlock + 1} / 12 - +
- + Head {attentionHead + 1} / 8 - +
-
+
OUTPUT PROBABILITIES
-
LayerNorm · MLP · softmax
+
single core pass · temperature 1
→ @@ -343,7 +346,7 @@
Model provenance - Mapped official TabICLv2 weights; SHA-256 verified before loading. The 12-row Iris trace is an out-of-regime illustration because the documented pretraining range starts at 300 rows. + Quantized official TabICLv2 weights; SHA-256 verified. Fixed single core pass, not the eight-view ensemble. The 12-row context is below the documented pretraining range; this illustrates computation, not model quality.
diff --git a/materials/website/index.html b/materials/website/index.html index 0637646..d44b8e7 100644 --- a/materials/website/index.html +++ b/materials/website/index.html @@ -587,7 +587,7 @@

Use the context to predict the hidden answer

mini visual · what survives? -
reveal 1→update θ→discard D¹
+
reveal 1→update θ→discard D¹
θ′ continues to the next task