Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions bun.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions packages/studio/package.json
Original file line number Diff line number Diff line change
Expand Up @@ -75,6 +75,7 @@
"@mcp-b/global": "^5.0.1",
"@phosphor-icons/react": "^2.1.10",
"@tanstack/react-virtual": "^3.14.6",
"acorn": "^8.17.0",
"bpm-detective": "^2.0.5",
"dompurify": "^3.2.4",
"gsap": "^3.13.0",
Expand Down
72 changes: 71 additions & 1 deletion packages/studio/src/captions/parser.test.ts
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
// @vitest-environment node
import { describe, it, expect } from "vitest";
import { describe, it, expect, vi } from "vitest";
import { extractTranscript, buildCaptionModel, TranscriptWord } from "./parser";
import { DEFAULT_STYLE, DEFAULT_CONTAINER, DEFAULT_ANIMATION_SET } from "./types";

Expand Down Expand Up @@ -156,6 +156,76 @@ describe("extractTranscript", () => {
});
});

it.each([
["array delimiter", "'x ]; y'", "x ]; y"],
["trailing-comma text", "'literal ,] and ,}'", "literal ,] and ,}"],
["object-key text", "'{ text: example }'", "{ text: example }"],
["escaped quotes", String.raw`'It\'s "quoted"'`, `It's "quoted"`],
["escaped double quotes", String.raw`'Say \"hello\"'`, 'Say "hello"'],
["escaped backslash", String.raw`'C:\\captions\\'`, "C:\\captions\\"],
["hex and Unicode escapes", String.raw`'\x41\u0042\u{1D11E}'`, "AB\u{1D11E}"],
["control escapes", String.raw`'one\ntwo\tthree'`, "one\ntwo\tthree"],
])("preserves %s in JavaScript strings", (_name, literal, text) => {
const source = `<script>const TRANSCRIPT = [{ id: 'word-a', text: ${literal}, start: 0.25, end: 1.5 }];</script>`;
expect(extractTranscript(source)).toEqual([{ id: "word-a", text, start: 0.25, end: 1.5 }]);
});

it("preserves JSON values and filters malformed word entries", () => {
const source = `const TRANSCRIPT = [
null, 42, {"text":"missing timing"}, {"text":"bad timing","start":"0","end":1},
{"id":7,"text":"valid","start":-0.25,"end":1e1,"extra":{"items":[true,false,null]}}
];`;
expect(extractTranscript(source)).toEqual([{ text: "valid", start: -0.25, end: 10 }]);
});

it("accepts comments around static values", () => {
expect(
extractTranscript(`const TRANSCRIPT = [
// Delimiter in a comment: ];
{ text: 'hello', start: 0, end: 1, },
] /* trailing comment */;`),
).toEqual([{ text: "hello", start: 0, end: 1 }]);
});

it.each([
"[{ text: 'unterminated, start: 0, end: 1 }];",
"[{ text: 'hello', start: 0, end: 1 };",
String.raw`[{ text: '\xZZ', start: 0, end: 1 }];`,
"[{ text: 'hello', start: 0, end: 1 }] garbage;",
"[{ text: 'hello', start: 0, end: 1 },,];",
"[{ text: getText(), start: 0, end: 1 }];",
"[{ text: 'hello', start: offset, end: 1 }];",
"[{ text: 'hello', start: 1 + 2, end: 4 }];",
"[{ text: 'hello', start: -getTime(), end: 1 }];",
"[{ text: 'hello', start: 0, end: 1, extra: /pattern/ }];",
"[{ text: 'hello', start: 0, end: 1, extra: 1n }];",
"[{ get text() { return 'hello'; }, start: 0, end: 1 }];",
"[{ text: 'hello', start: 0, end: 1, extra() {} }];",
"[{ ['text']: 'hello', start: 0, end: 1 }];",
"[{ text, start: 0, end: 1 }];",
"[{ ...word, text: 'hello', start: 0, end: 1 }];",
"[...words];",
"[{ text: `hello ${getText()}`, start: 0, end: 1 }];",
"[{ text: 'hello', start: 0, end: 1, extra: { run: getText() } }];",
"[{ text: 'hello', start: 0, end: 1 }].map(transform);",
"[{ text: 'hello', start: 0, end: 1 }], run();",
])("rejects malformed or executable initializers: %s", (initializer) => {
expect(extractTranscript(`const TRANSCRIPT = ${initializer}`)).toEqual([]);
});

it("never runs code while extracting a transcript", () => {
const getCaptionText = vi.fn(() => "hello");
vi.stubGlobal("getCaptionText", getCaptionText);
try {
expect(
extractTranscript("const TRANSCRIPT = [{ text: getCaptionText(), start: 0, end: 1 }];"),
).toEqual([]);
expect(getCaptionText).not.toHaveBeenCalled();
} finally {
vi.unstubAllGlobals();
}
});

describe("real-world source samples", () => {
it("handles a realistic production-style TRANSCRIPT block with many words", () => {
const source = `
Expand Down
118 changes: 76 additions & 42 deletions packages/studio/src/captions/parser.ts
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,14 @@
// Parses a caption composition's JavaScript source to extract the transcript word array,
// and builds a CaptionModel from a TranscriptWord array.

import {
parseExpressionAt,
tokenizer,
type Expression,
type Property,
type SpreadElement,
type UnaryExpression,
} from "acorn";
import {
CaptionModel,
CaptionSegment,
Expand Down Expand Up @@ -102,24 +110,26 @@ export function buildCaptionModel(
* Looks for `const TRANSCRIPT = [...]` or `const script = [...]` (also let/var)
* and parses each `{ text, start, end }` object into TranscriptWord objects.
*
* Supports static literals only; expressions are never evaluated.
* Returns an empty array if no transcript is found or if parsing fails.
*/
export function extractTranscript(source: string): TranscriptWord[] {
// Match: (const|let|var) (TRANSCRIPT|script) = [...]
// The array may span multiple lines and contain trailing commas.
// The lazy [\s\S]*? anchors on the first `];` — assumes transcript word
// text never contains a literal `];` string (safe for speech transcripts).
const varPattern = /(?:const|let|var)\s+(?:TRANSCRIPT|script)\s*=\s*(\[[\s\S]*?\]);/;
const match = source.match(varPattern);
// Locate the initializer in either JavaScript or a complete HTML composition.
// Acorn owns its boundary so delimiters inside strings remain literal text.
const varPattern = /\b(?:const|let|var)\s+(?:TRANSCRIPT|script)\s*=\s*(?=\[)/;
const match = varPattern.exec(source);

if (!match) {
return [];
}

const arrayLiteral = match[1];

try {
return parseTranscriptArray(arrayLiteral);
const options = { ecmaVersion: "latest" } as const;
const expression = parseExpressionAt(source, match.index + match[0].length, options);
if (expression.type !== "ArrayExpression") return [];
const nextToken = tokenizer(source.slice(expression.end), options).getToken();
if (nextToken.type.label !== ";") return [];
return parseTranscriptArray(readStaticValue(expression));
} catch {
return [];
}
Expand Down Expand Up @@ -262,51 +272,75 @@ export function parseCaptionComposition(
return model;
}

/**
* Parses a JS array literal containing `{ text, start, end }` objects.
*
* Handles:
* - Double-quoted and single-quoted string values
* - Trailing commas after the last element or property
* - Unquoted property keys (standard JS object literal syntax)
* - Numeric values for start/end
*/
function parseTranscriptArray(arrayLiteral: string): TranscriptWord[] {
// Try parsing as-is first (handles already-valid JSON)
let parsed: unknown;
try {
parsed = JSON.parse(arrayLiteral);
} catch {
// Not valid JSON — normalize single quotes, unquoted keys, trailing commas
let normalized = arrayLiteral;
normalized = normalized.replace(/'((?:[^'\\]|\\.)*)'/g, (_match, inner) => {
const escaped = inner.replace(/\\'/g, "'").replace(/"/g, '\\"');
return `"${escaped}"`;
});
normalized = normalized.replace(/([{,]\s*)([a-zA-Z_$][a-zA-Z0-9_$]*)\s*:/g, '$1"$2":');
normalized = normalized.replace(/,(\s*[}\]])/g, "$1");
parsed = JSON.parse(normalized);
/** Decode data literals only, including metadata that is not used by captions. */
function readStaticValue(node: Expression | SpreadElement | null): unknown {
switch (node?.type) {
case "Literal":
// Acorn's remaining literals are strings, numbers, booleans, or null.
if ("regex" in node || "bigint" in node) break;
return node.value;
case "UnaryExpression":
return readSignedNumber(node);
case "ArrayExpression":
return node.elements.map(readStaticValue);
case "ObjectExpression":
return Object.fromEntries(node.properties.map(readStaticProperty));
}
throw new SyntaxError("Transcript values must be static literals");
}

function readSignedNumber(node: UnaryExpression): number {
if (
(node.operator !== "-" && node.operator !== "+") ||
node.argument.type !== "Literal" ||
typeof node.argument.value !== "number"
) {
throw new SyntaxError("Transcript unary expressions must be signed numeric literals");
}
return node.operator === "-" ? -node.argument.value : node.argument.value;
}

function readStaticProperty(property: Property | SpreadElement): [string | number, unknown] {
if (
property.type !== "Property" ||
property.kind !== "init" ||
property.method ||
property.shorthand ||
property.computed
) {
throw new SyntaxError("Transcript properties must be static data");
}
const key =
property.key.type === "Identifier" ? property.key.name : readStaticValue(property.key);
if (typeof key !== "string" && typeof key !== "number") {
throw new SyntaxError("Transcript property keys must be names or literals");
}
return [key, readStaticValue(property.value)];
}

function parseTranscriptArray(parsed: unknown): TranscriptWord[] {
if (!Array.isArray(parsed)) {
return [];
}

const words: TranscriptWord[] = [];
for (const item of parsed) {
const items: unknown[] = parsed;
for (const item of items) {
if (
item !== null &&
typeof item === "object" &&
typeof (item as Record<string, unknown>).text === "string" &&
typeof (item as Record<string, unknown>).start === "number" &&
typeof (item as Record<string, unknown>).end === "number"
"text" in item &&
typeof item.text === "string" &&
"start" in item &&
typeof item.start === "number" &&
"end" in item &&
typeof item.end === "number"
) {
const entry = item as Record<string, unknown>;
words.push({
...(typeof entry.id === "string" ? { id: entry.id } : {}),
text: entry.text as string,
start: entry.start as number,
end: entry.end as number,
...("id" in item && typeof item.id === "string" ? { id: item.id } : {}),
text: item.text,
start: item.start,
end: item.end,
});
}
}
Expand Down
Loading