Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
124 changes: 124 additions & 0 deletions apps/web/__tests__/integration/live-transcribe-diarization.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,124 @@
import { Effect, Option } from "effect";
import { beforeEach, describe, expect, it, vi } from "vitest";
import { createEmptyLiveTranscript } from "@/lib/live-transcribe-core";

const mocks = vi.hoisted(() => ({
queue: vi.fn(),
objects: new Map<string, string>(),
writes: [] as string[],
}));
const video = {
id: "live-video",
ownerId: "live-owner",
source: { type: "desktopSegments" },
transcriptionStatus: null,
settings: null,
};
vi.mock("@cap/env", () => ({
serverEnv: () => ({ ASSEMBLY_API_KEY: "test-key" }),
}));
vi.mock("@cap/database/schema", () => ({
videos: {
id: "video-id",
ownerId: "owner-id",
metadata: "metadata",
updatedAt: "updated-at",
},
organizations: { id: "org-id", settings: "settings" },
users: { id: "user-id" },
}));
vi.mock("@cap/database", () => ({
db: () => ({
select: () => ({
from: () => ({
leftJoin: () => ({ where: async () => [{ video, orgSettings: null }] }),
where: async () => [video],
}),
}),
update: () => ({ set: () => ({ where: async () => [] }) }),
}),
}));
vi.mock("@cap/web-backend/src/Storage/index", () => ({
Storage: {
getAccessForVideo: () =>
Effect.succeed([
{
getObject: (key: string) =>
Effect.succeed(Option.fromNullable(mocks.objects.get(key))),
putObject: (key: string, value: string) =>
Effect.sync(() => {
mocks.writes.push(key);
mocks.objects.set(key, value);
}),
},
]),
},
}));
vi.mock("@/lib/video-storage", () => ({ decodeStorageVideo: () => ({}) }));
vi.mock("@/lib/workflow-runtime", () => ({
runWorkflowPromise: Effect.runPromise,
}));
vi.mock("@/lib/transcribe", () => ({ transcribeVideo: mocks.queue }));
vi.mock("@/lib/ai-generation-entitlement", () => ({
isAiGenerationEnabledForUser: () => false,
}));

const artifactKey = "live-owner/live-video/transcription.live.json";
beforeEach(() => {
mocks.objects.clear();
mocks.writes.length = 0;
mocks.queue.mockResolvedValue({ success: true, message: "Queued" });
mocks.objects.set(
artifactKey,
JSON.stringify({
...createEmptyLiveTranscript("2026-09-08T00:00:00.000Z"),
lastAudioSegmentIndex: 2,
transcribedDurationMs: 4000,
}),
);
mocks.objects.set(
"live-owner/live-video/segments/manifest.json",
JSON.stringify({
version: 5,
video_init_uploaded: true,
audio_init_uploaded: true,
video_segments: [],
audio_segments: [
{ index: 1, duration: 2 },
{ index: 2, duration: 2 },
],
is_complete: true,
}),
);
});

describe("live recording diarization handoff", () => {
it("queues a full recording pass even when provisional chunks cover every segment", async () => {
const { liveTranscribeWorkflow } = await import(
"@/workflows/live-transcribe"
);
await liveTranscribeWorkflow({ videoId: video.id, userId: video.ownerId });
expect(mocks.queue).toHaveBeenCalledExactlyOnceWith(
video.id,
video.ownerId,
false,
{ earlyFromSegments: true },
);
expect(mocks.writes).toEqual([artifactKey]);
expect(JSON.parse(mocks.objects.get(artifactKey) ?? "{}").state).toBe(
"complete",
);
});
it("surfaces queue failures so the durable workflow retries the final transcription", async () => {
mocks.queue.mockResolvedValue({
success: false,
message: "Queue unavailable",
});
const { liveTranscribeWorkflow } = await import(
"@/workflows/live-transcribe"
);
await expect(
liveTranscribeWorkflow({ videoId: video.id, userId: video.ownerId }),
).rejects.toThrow("Queue unavailable");
});
});
6 changes: 6 additions & 0 deletions apps/web/__tests__/integration/transcribe-workflow.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -209,6 +209,9 @@ describe("transcribeVideoWorkflow", () => {

expect(result.success).toBe(true);
expect(mocks.transcribe).toHaveBeenCalledTimes(1);
expect(mocks.transcribe).toHaveBeenCalledWith(
expect.objectContaining({ speaker_labels: true }),
);
expect(mocks.transcribe.mock.calls[0]?.[0]).toMatchObject({
disfluencies: true,
speech_models: ["universal-3-5-pro", "universal-2"],
Expand Down Expand Up @@ -270,6 +273,9 @@ describe("transcribeVideoWorkflow", () => {
message: "Video has no spoken audio - skipped transcription",
});
expect(mocks.transcribe).toHaveBeenCalledTimes(1);
expect(mocks.transcribe).toHaveBeenCalledWith(
expect.objectContaining({ speaker_labels: true }),
);
expect(mocks.updates).toContainEqual({ transcriptionStatus: "NO_AUDIO" });
expect(mocks.updates).not.toContainEqual({ transcriptionStatus: "ERROR" });
expect(mocks.startAiGeneration).not.toHaveBeenCalled();
Expand Down
4 changes: 2 additions & 2 deletions apps/web/__tests__/unit/agent-api.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -77,14 +77,14 @@ describe("agent API normalization", () => {
expect(agentTranscriptRevision(rendered)).toMatch(/^[a-f0-9]{64}$/);
});

it("removes complete and unterminated VTT tags", () => {
it("removes complete tags and keeps unfinished literal text", () => {
const cues = parseAgentVtt(
"WEBVTT\n\n1\n00:00:00.000 --> 00:00:01.000\nSafe <script>alert(1)</script> text\n\n2\n00:00:01.000 --> 00:00:02.000\nBefore <script\n",
);

expect(cues).toEqual([
{ startMs: 0, endMs: 1_000, text: "Safe alert(1) text" },
{ startMs: 1_000, endMs: 2_000, text: "Before" },
{ startMs: 1_000, endMs: 2_000, text: "Before <script" },
]);
});

Expand Down
16 changes: 14 additions & 2 deletions apps/web/__tests__/unit/caption-cues.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -20,9 +20,21 @@ describe("getActiveCaptionText", () => {
it("uses the latest active cue when cues overlap", () => {
const activeCues = createCueList([
{ startTime: 0, text: "First caption" },
{ startTime: 3.199, text: "<v Speaker>Second caption</v>" },
{ startTime: 3.199, text: "<v Speaker B>Second &amp; final caption</v>" },
]);

expect(getActiveCaptionText(activeCues)).toBe("Second caption");
expect(getActiveCaptionText(activeCues)).toBe(
"Speaker B: Second & final caption",
);
});

it("preserves legacy literal angle brackets alongside voice markup", () => {
expect(
getActiveCaptionText(
createCueList([
{ startTime: 0, text: "<v Speaker A>2 < 3 and 4 > 1</v>" },
]),
),
).toBe("Speaker A: 2 < 3 and 4 > 1");
});
});
175 changes: 175 additions & 0 deletions apps/web/__tests__/unit/diarization.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,175 @@
import { describe, expect, it } from "vitest";
import { formatTranscriptAsVTT } from "@/app/s/[videoId]/_components/utils/transcript-utils";
import {
createEditTranscript,
editTranscriptWordsToCaptionVtt,
groupEditTranscriptWords,
parseEditTranscript,
remapEditTranscriptThroughSpec,
serializeEditTranscript,
} from "@/lib/edit-transcript";
import { formatTranscriptAsParagraphs } from "@/lib/transcript-text";
import {
formatVttCueText,
parseVTT,
parseVttCueText,
updateVttEntryText,
} from "@/lib/transcript-vtt";

const transcript = createEditTranscript(
{
words: [
{ text: "Hello", start: 100, end: 300, speaker: "A" },
{ text: "there", start: 300, end: 600, speaker: "A" },
{ text: "Hi", start: 650, end: 800, speaker: "B" },
{ text: "again", start: 850, end: 1000, speaker: "A" },
{ text: "unknown", start: 1100, end: 1300, speaker: null },
],
},
2000,
);

describe("speaker diarization", () => {
it("splits captions on every speaker transition without needing punctuation or silence", () => {
const cues = parseVTT(editTranscriptWordsToCaptionVtt(transcript.words));
expect(
cues.map(({ text, speaker, startTime, endTime }) => ({
text,
speaker,
startTime,
endTime,
})),
).toEqual([
{ text: "Hello there", speaker: "A", startTime: 0.1, endTime: 0.6 },
{ text: "Hi", speaker: "B", startTime: 0.65, endTime: 0.8 },
{ text: "again", speaker: "A", startTime: 0.85, endTime: 1 },
{ text: "unknown", speaker: null, startTime: 1.1, endTime: 1.3 },
]);
});

it("preserves labels through storage, video cuts, caption regeneration, and download", () => {
const stored = parseEditTranscript(serializeEditTranscript(transcript));
expect(stored).not.toBeNull();
if (!stored) throw new Error("Missing transcript");
const edited = remapEditTranscriptThroughSpec(stored, {
version: 1,
sourceDuration: 2,
keepRanges: [{ start: 0.6, end: 2 }],
});
const cues = parseVTT(editTranscriptWordsToCaptionVtt(edited.words));
expect(cues[0]).toMatchObject({
text: "Hi",
speaker: "B",
startTime: 0.05,
});
expect(parseVTT(formatTranscriptAsVTT(cues))).toEqual(cues);
});

it("shows separate editor groups and text paragraphs for speakers and unknown speech", () => {
expect(
groupEditTranscriptWords(transcript.words).map(
({ startIndex, endIndex }) => [startIndex, endIndex],
),
).toEqual([
[0, 1],
[2, 2],
[3, 3],
[4, 4],
]);
expect(
formatTranscriptAsParagraphs(
parseVTT(editTranscriptWordsToCaptionVtt(transcript.words)),
),
).toBe(
"Speaker A: Hello there\n\nSpeaker B: Hi\n\nSpeaker A: again\n\nunknown",
);
});

it("preserves the voice when editing spoken text and escapes markup", () => {
const vtt = editTranscriptWordsToCaptionVtt(transcript.words);
const updated = updateVttEntryText(vtt, 2, "Yes <script> & no");
expect(updated.updated).toBe(true);
expect(updated.content).toContain(
"<v Speaker B>Yes &lt;script&gt; &amp; no</v>",
);
expect(parseVTT(updated.content)[1]).toMatchObject({
speaker: "B",
text: "Yes <script> & no",
startTime: 0.65,
endTime: 0.8,
});
});

it("handles multiline voice cues, legacy plain cues, and escaped labels", () => {
const cues = parseVTT(
"WEBVTT\r\n\r\n1\r\n00:00:00.125 --> 00:00:01.500\r\n<v Speaker A>Hello\r\nthere</v>\r\n\r\n2\r\n00:00:01.500 --> 00:00:02.000\r\nLegacy text\r\n",
);
expect(cues[0]).toMatchObject({
text: "Hello there",
speaker: "A",
startTime: 0.125,
endTime: 1.5,
});
expect(cues[1]).toMatchObject({ text: "Legacy text", speaker: null });
expect(parseVttCueText(formatVttCueText("2 < 3 & 4 > 1", "A & B"))).toEqual(
{ text: "2 < 3 & 4 > 1", speaker: "A & B" },
);
});

it("preserves literal angle brackets in legacy cues through copying and exports", () => {
const cues = parseVTT(
"WEBVTT\n\n1\n00:00:00.000 --> 00:00:02.000\n2 < 3 and 4 > 1\n\n2\n00:00:02.000 --> 00:00:03.000\nThe answer is < 5\n",
);
expect(cues.map(({ text }) => text)).toEqual([
"2 < 3 and 4 > 1",
"The answer is < 5",
]);
expect(formatTranscriptAsParagraphs(cues)).toBe(
"2 < 3 and 4 > 1 The answer is < 5",
);
expect(parseVTT(formatTranscriptAsVTT(cues))).toEqual(cues);
});

it("strips recognized WebVTT markup and timestamps without stripping literal text", () => {
expect(
parseVttCueText(
"<v.class Speaker A><b>2 < 3</b> <c.green>and</c> <i>4 > 1</i> <00:00:01.000><lang en><u>five</u></lang> <ruby>six<rt>6</rt></ruby></v>",
),
).toEqual({ text: "2 < 3 and 4 > 1 five six6", speaker: "A" });
expect(parseVttCueText("Literal <value> &lt;b&gt; and <b")).toEqual({
text: "Literal <value> <b> and <b",
speaker: null,
});
});
});

it("keeps speaker metadata and literal text through the agent transcript API", async () => {
const { parseAgentVtt, renderAgentVtt } = await import("@/lib/agent-api");
const cues = [
{ startMs: 125, endMs: 500, text: "R&D < planning", speaker: "B" },
];
expect(parseAgentVtt(renderAgentVtt(cues))).toEqual(cues);
expect(
parseAgentVtt(
"WEBVTT\n\n1\n00:00:00.125 --> 00:00:00.500\n<v Speaker B>R&D < planning</v>\n",
),
).toEqual(cues);
});

it("keeps escaped angle-bracket speech through the agent transcript API", async () => {
const { parseAgentVtt, renderAgentVtt } = await import("@/lib/agent-api");
const cues = [
{ startMs: 0, endMs: 1_000, text: "Type <value> then <b", speaker: "A" },
];
expect(parseAgentVtt(renderAgentVtt(cues))).toEqual(cues);
expect(
parseAgentVtt(
"WEBVTT\n\n1\n00:00:00.000 --> 00:00:01.000\n<v Speaker A><c.green>Type &lt;value&gt;</c> <00:00:00.500>then &lt;b</v>\n",
),
).toEqual(cues);
expect(
parseAgentVtt(
"WEBVTT\n\n1\n00:00:00.000 --> 00:00:01.000\n<v Speaker A>Type &lt;value&gt; then <b</v>\n",
),
).toEqual(cues);
});
Loading