From 203897b30cbf16d40757167388532304462feb6f Mon Sep 17 00:00:00 2001 From: pavzagor Date: Tue, 8 Sep 2026 14:50:11 +0300 Subject: [PATCH 01/11] feat: show AssemblyAI speaker diarization in transcripts --- .../live-transcribe-diarization.test.ts | 124 +++++++ .../integration/transcribe-workflow.test.ts | 6 + apps/web/__tests__/unit/caption-cues.test.ts | 6 +- apps/web/__tests__/unit/diarization.test.ts | 126 ++++++++ .../unit/live-transcribe-core.test.ts | 91 ------ .../unit/transcribe-language.test.ts | 8 + .../actions/videos/translate-transcript.ts | 3 +- apps/web/app/api/v1/[...route]/route.ts | 9 +- .../s/[videoId]/_components/caption-cues.ts | 6 +- .../[videoId]/_components/tabs/Transcript.tsx | 137 ++------ .../_components/utils/transcript-utils.ts | 112 +------ .../s/[videoId]/edit/TranscriptSidebar.tsx | 5 + apps/web/lib/agent-api.ts | 30 +- apps/web/lib/assemblyai.ts | 2 + apps/web/lib/edit-transcript.ts | 6 +- apps/web/lib/live-transcribe-core.ts | 69 +--- apps/web/lib/transcribe-utils.ts | 12 +- apps/web/lib/transcribe.ts | 5 +- apps/web/lib/transcript-text.ts | 10 +- apps/web/lib/transcript-vtt.ts | 179 ++++++++++- apps/web/workflows/live-transcribe.ts | 302 ++---------------- packages/web-domain/src/Agent.ts | 1 + 22 files changed, 563 insertions(+), 686 deletions(-) create mode 100644 apps/web/__tests__/integration/live-transcribe-diarization.test.ts create mode 100644 apps/web/__tests__/unit/diarization.test.ts diff --git a/apps/web/__tests__/integration/live-transcribe-diarization.test.ts b/apps/web/__tests__/integration/live-transcribe-diarization.test.ts new file mode 100644 index 00000000000..a8f77c37abb --- /dev/null +++ b/apps/web/__tests__/integration/live-transcribe-diarization.test.ts @@ -0,0 +1,124 @@ +import { Effect, Option } from "effect"; +import { beforeEach, describe, expect, it, vi } from "vitest"; +import { createEmptyLiveTranscript } from "@/lib/live-transcribe-core"; + +const mocks = vi.hoisted(() => ({ + queue: vi.fn(), + objects: new Map(), + writes: [] as string[], +})); +const video = { + id: "live-video", + ownerId: "live-owner", + source: { type: "desktopSegments" }, + transcriptionStatus: null, + settings: null, +}; +vi.mock("@cap/env", () => ({ + serverEnv: () => ({ ASSEMBLY_API_KEY: "test-key" }), +})); +vi.mock("@cap/database/schema", () => ({ + videos: { + id: "video-id", + ownerId: "owner-id", + metadata: "metadata", + updatedAt: "updated-at", + }, + organizations: { id: "org-id", settings: "settings" }, + users: { id: "user-id" }, +})); +vi.mock("@cap/database", () => ({ + db: () => ({ + select: () => ({ + from: () => ({ + leftJoin: () => ({ where: async () => [{ video, orgSettings: null }] }), + where: async () => [video], + }), + }), + update: () => ({ set: () => ({ where: async () => [] }) }), + }), +})); +vi.mock("@cap/web-backend/src/Storage/index", () => ({ + Storage: { + getAccessForVideo: () => + Effect.succeed([ + { + getObject: (key: string) => + Effect.succeed(Option.fromNullable(mocks.objects.get(key))), + putObject: (key: string, value: string) => + Effect.sync(() => { + mocks.writes.push(key); + mocks.objects.set(key, value); + }), + }, + ]), + }, +})); +vi.mock("@/lib/video-storage", () => ({ decodeStorageVideo: () => ({}) })); +vi.mock("@/lib/workflow-runtime", () => ({ + runWorkflowPromise: Effect.runPromise, +})); +vi.mock("@/lib/transcribe", () => ({ transcribeVideo: mocks.queue })); +vi.mock("@/lib/ai-generation-entitlement", () => ({ + isAiGenerationEnabledForUser: () => false, +})); + +const artifactKey = "live-owner/live-video/transcription.live.json"; +beforeEach(() => { + mocks.objects.clear(); + mocks.writes.length = 0; + mocks.queue.mockResolvedValue({ success: true, message: "Queued" }); + mocks.objects.set( + artifactKey, + JSON.stringify({ + ...createEmptyLiveTranscript("2026-09-08T00:00:00.000Z"), + lastAudioSegmentIndex: 2, + transcribedDurationMs: 4000, + }), + ); + mocks.objects.set( + "live-owner/live-video/segments/manifest.json", + JSON.stringify({ + version: 5, + video_init_uploaded: true, + audio_init_uploaded: true, + video_segments: [], + audio_segments: [ + { index: 1, duration: 2 }, + { index: 2, duration: 2 }, + ], + is_complete: true, + }), + ); +}); + +describe("live recording diarization handoff", () => { + it("queues a full recording pass even when provisional chunks cover every segment", async () => { + const { liveTranscribeWorkflow } = await import( + "@/workflows/live-transcribe" + ); + await liveTranscribeWorkflow({ videoId: video.id, userId: video.ownerId }); + expect(mocks.queue).toHaveBeenCalledExactlyOnceWith( + video.id, + video.ownerId, + false, + { earlyFromSegments: true }, + ); + expect(mocks.writes).toEqual([artifactKey]); + expect(JSON.parse(mocks.objects.get(artifactKey) ?? "{}").state).toBe( + "complete", + ); + }); + it("surfaces queue failures so the durable workflow retries the final transcription", async () => { + mocks.queue.mockResolvedValue({ + success: false, + message: "Queue unavailable", + }); + const { liveTranscribeWorkflow } = await import( + "@/workflows/live-transcribe" + ); + await expect( + liveTranscribeWorkflow({ videoId: video.id, userId: video.ownerId }), + ).rejects.toThrow("Queue unavailable"); + }); +}); diff --git a/apps/web/__tests__/integration/transcribe-workflow.test.ts b/apps/web/__tests__/integration/transcribe-workflow.test.ts index 3ca5d49b93e..110a20b7277 100644 --- a/apps/web/__tests__/integration/transcribe-workflow.test.ts +++ b/apps/web/__tests__/integration/transcribe-workflow.test.ts @@ -209,6 +209,9 @@ describe("transcribeVideoWorkflow", () => { expect(result.success).toBe(true); expect(mocks.transcribe).toHaveBeenCalledTimes(1); + expect(mocks.transcribe).toHaveBeenCalledWith( + expect.objectContaining({ speaker_labels: true }), + ); expect(mocks.transcribe.mock.calls[0]?.[0]).toMatchObject({ disfluencies: true, speech_models: ["universal-3-5-pro", "universal-2"], @@ -270,6 +273,9 @@ describe("transcribeVideoWorkflow", () => { message: "Video has no spoken audio - skipped transcription", }); expect(mocks.transcribe).toHaveBeenCalledTimes(1); + expect(mocks.transcribe).toHaveBeenCalledWith( + expect.objectContaining({ speaker_labels: true }), + ); expect(mocks.updates).toContainEqual({ transcriptionStatus: "NO_AUDIO" }); expect(mocks.updates).not.toContainEqual({ transcriptionStatus: "ERROR" }); expect(mocks.startAiGeneration).not.toHaveBeenCalled(); diff --git a/apps/web/__tests__/unit/caption-cues.test.ts b/apps/web/__tests__/unit/caption-cues.test.ts index daecbe46f2e..6afc5522ba7 100644 --- a/apps/web/__tests__/unit/caption-cues.test.ts +++ b/apps/web/__tests__/unit/caption-cues.test.ts @@ -20,9 +20,11 @@ describe("getActiveCaptionText", () => { it("uses the latest active cue when cues overlap", () => { const activeCues = createCueList([ { startTime: 0, text: "First caption" }, - { startTime: 3.199, text: "Second caption" }, + { startTime: 3.199, text: "Second & final caption" }, ]); - expect(getActiveCaptionText(activeCues)).toBe("Second caption"); + expect(getActiveCaptionText(activeCues)).toBe( + "Speaker B: Second & final caption", + ); }); }); diff --git a/apps/web/__tests__/unit/diarization.test.ts b/apps/web/__tests__/unit/diarization.test.ts new file mode 100644 index 00000000000..ada060ab265 --- /dev/null +++ b/apps/web/__tests__/unit/diarization.test.ts @@ -0,0 +1,126 @@ +import { describe, expect, it } from "vitest"; +import { formatTranscriptAsVTT } from "@/app/s/[videoId]/_components/utils/transcript-utils"; +import { + createEditTranscript, + editTranscriptWordsToCaptionVtt, + groupEditTranscriptWords, + parseEditTranscript, + remapEditTranscriptThroughSpec, + serializeEditTranscript, +} from "@/lib/edit-transcript"; +import { formatTranscriptAsParagraphs } from "@/lib/transcript-text"; +import { + formatVttCueText, + parseVTT, + parseVttCueText, + updateVttEntryText, +} from "@/lib/transcript-vtt"; + +const transcript = createEditTranscript( + { + words: [ + { text: "Hello", start: 100, end: 300, speaker: "A" }, + { text: "there", start: 300, end: 600, speaker: "A" }, + { text: "Hi", start: 650, end: 800, speaker: "B" }, + { text: "again", start: 850, end: 1000, speaker: "A" }, + { text: "unknown", start: 1100, end: 1300, speaker: null }, + ], + }, + 2000, +); + +describe("speaker diarization", () => { + it("splits captions on every speaker transition without needing punctuation or silence", () => { + const cues = parseVTT(editTranscriptWordsToCaptionVtt(transcript.words)); + expect( + cues.map(({ text, speaker, startTime, endTime }) => ({ + text, + speaker, + startTime, + endTime, + })), + ).toEqual([ + { text: "Hello there", speaker: "A", startTime: 0.1, endTime: 0.6 }, + { text: "Hi", speaker: "B", startTime: 0.65, endTime: 0.8 }, + { text: "again", speaker: "A", startTime: 0.85, endTime: 1 }, + { text: "unknown", speaker: null, startTime: 1.1, endTime: 1.3 }, + ]); + }); + + it("preserves labels through storage, video cuts, caption regeneration, and download", () => { + const stored = parseEditTranscript(serializeEditTranscript(transcript)); + expect(stored).not.toBeNull(); + if (!stored) throw new Error("Missing transcript"); + const edited = remapEditTranscriptThroughSpec(stored, { + version: 1, + sourceDuration: 2, + keepRanges: [{ start: 0.6, end: 2 }], + }); + const cues = parseVTT(editTranscriptWordsToCaptionVtt(edited.words)); + expect(cues[0]).toMatchObject({ + text: "Hi", + speaker: "B", + startTime: 0.05, + }); + expect(parseVTT(formatTranscriptAsVTT(cues))).toEqual(cues); + }); + + it("shows separate editor groups and text paragraphs for speakers and unknown speech", () => { + expect( + groupEditTranscriptWords(transcript.words).map( + ({ startIndex, endIndex }) => [startIndex, endIndex], + ), + ).toEqual([ + [0, 1], + [2, 2], + [3, 3], + [4, 4], + ]); + expect( + formatTranscriptAsParagraphs( + parseVTT(editTranscriptWordsToCaptionVtt(transcript.words)), + ), + ).toBe( + "Speaker A: Hello there\n\nSpeaker B: Hi\n\nSpeaker A: again\n\nunknown", + ); + }); + + it("preserves the voice when editing spoken text and escapes markup", () => { + const vtt = editTranscriptWordsToCaptionVtt(transcript.words); + const updated = updateVttEntryText(vtt, 2, "Yes text\n\n2\n00:00:01.000 --> 00:00:02.000\nBefore 00:00:01.000\nType <value> <00:00:00.500>then <b\n", ), ).toEqual(cues); + expect( + parseAgentVtt( + "WEBVTT\n\n1\n00:00:00.000 --> 00:00:01.000\nType <value> then \n", + ), + ).toEqual(cues); }); diff --git a/apps/web/lib/agent-api.ts b/apps/web/lib/agent-api.ts index f3fd2b27ca7..63e180856d2 100644 --- a/apps/web/lib/agent-api.ts +++ b/apps/web/lib/agent-api.ts @@ -112,7 +112,7 @@ const parseVttTimestamp = (value: string) => { }; const stripMarkupTags = (value: string) => - value.replace(/<[A-Za-z/!?][^<>]*(?:>|$)/g, "").trim(); + value.replace(/<[A-Za-z/!?][^<>]*>/g, "").trim(); export const parseAgentVtt = ( vtt: string, From d0562e3f16585361061c8d348c3492118f1373bf Mon Sep 17 00:00:00 2001 From: pavzagor Date: Wed, 9 Sep 2026 17:38:05 +0300 Subject: [PATCH 08/11] fix: group transcript captions into readable sentences --- .../unit/transcript-sentences-ui.test.ts | 148 ++++++++++++++++++ .../unit/transcript-sentences.test.ts | 119 ++++++++++++++ .../[videoId]/_components/tabs/Transcript.tsx | 39 ++++- apps/web/lib/transcript-sentences.ts | 46 ++++++ 4 files changed, 346 insertions(+), 6 deletions(-) create mode 100644 apps/web/__tests__/unit/transcript-sentences-ui.test.ts create mode 100644 apps/web/__tests__/unit/transcript-sentences.test.ts create mode 100644 apps/web/lib/transcript-sentences.ts diff --git a/apps/web/__tests__/unit/transcript-sentences-ui.test.ts b/apps/web/__tests__/unit/transcript-sentences-ui.test.ts new file mode 100644 index 00000000000..42aa680911f --- /dev/null +++ b/apps/web/__tests__/unit/transcript-sentences-ui.test.ts @@ -0,0 +1,148 @@ +// @vitest-environment jsdom + +import { act, createElement } from "react"; +import { createRoot } from "react-dom/client"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import { Transcript } from "@/app/s/[videoId]/_components/tabs/Transcript"; +import type { VideoData } from "@/app/s/[videoId]/types"; + +const mocks = vi.hoisted(() => ({ + content: + "WEBVTT\n\n1\n00:00:00.125 --> 00:00:01.000\nFirst,\n\n2\n00:00:02.000 --> 00:00:03.000\nthen second.\n\n3\n00:00:03.100 --> 00:00:04.000\nYes.\n\n", + edit: vi.fn(async () => ({ success: true })), + copy: vi.fn(async () => {}), + language: "original", + live: false, +})); + +vi.mock("@cap/ui", () => ({ Button: "button" })); +vi.mock("@tanstack/react-query", () => ({ + useQueryClient: () => ({ invalidateQueries: vi.fn() }), + useMutation: () => ({}), +})); +vi.mock("hooks/use-transcript", () => ({ + useTranscript: () => ({ data: mocks.content, isLoading: false }), + useInvalidateTranscript: () => vi.fn(), +})); +vi.mock("hooks/use-live-transcript", () => ({ + useLiveTranscript: () => ({ + data: mocks.live + ? { kind: "ready", state: "active", content: mocks.content } + : undefined, + }), +})); +vi.mock("@/app/Layout/AuthContext", () => ({ + useCurrentUser: () => ({ id: "owner" }), +})); +vi.mock("@/actions/videos/edit-transcript", () => ({ + editTranscriptEntry: mocks.edit, +})); +vi.mock("@/app/s/[videoId]/_components/CaptionContext", () => ({ + useCaptionContext: () => ({ + selectedLanguage: mocks.language, + currentVttContent: mocks.content, + translatedVttContent: new Map(), + isTranslating: false, + }), +})); + +const data = { + id: "sentence-test", + owner: { id: "owner" }, + transcriptionStatus: "COMPLETE", + createdAt: new Date(), +} as unknown as VideoData; + +let container: HTMLDivElement; +let root: ReturnType; + +function button(label: string): HTMLButtonElement { + const match = Array.from(container.querySelectorAll("button")).find( + (element) => + element.textContent?.trim() === label || + element.getAttribute("aria-label") === label, + ); + if (!match) throw new Error(`Missing button: ${label}`); + return match; +} + +async function click(label: string) { + await act(async () => button(label).click()); +} + +describe("sentence transcript reading and editing", () => { + beforeEach(() => { + vi.stubGlobal("IS_REACT_ACT_ENVIRONMENT", true); + Object.defineProperty(navigator, "clipboard", { + configurable: true, + value: { writeText: mocks.copy }, + }); + mocks.edit.mockClear(); + mocks.copy.mockClear(); + mocks.language = "original"; + mocks.live = false; + container = document.createElement("div"); + document.body.append(container); + root = createRoot(container); + }); + + afterEach(async () => { + await act(async () => root.unmount()); + container.remove(); + vi.unstubAllGlobals(); + }); + + it("reads sentences, seeks precisely, and copies sentence timestamps", async () => { + const seek = vi.fn(); + await act(async () => + root.render(createElement(Transcript, { data, onSeek: seek })), + ); + await click("00:00Speaker AFirst, then second."); + expect(seek).toHaveBeenCalledWith(0.125); + expect( + container.querySelectorAll('[aria-label="Edit transcript entry"]'), + ).toHaveLength(0); + await click("Copy transcript"); + const copyOption = Array.from(container.querySelectorAll("button")).find( + (element) => element.textContent?.startsWith("With timestamps"), + ); + expect(copyOption).toBeDefined(); + await act(async () => copyOption?.click()); + expect(mocks.copy).toHaveBeenCalledWith( + "[00:00] Speaker A: First, then second.\n\n[00:03] Speaker B: Yes.", + ); + }); + + it("edits original cue IDs and text, never a merged sentence under one cue ID", async () => { + await act(async () => root.render(createElement(Transcript, { data }))); + await click("Edit transcript"); + const editButtons = container.querySelectorAll( + '[aria-label="Edit transcript entry"]', + ); + expect(editButtons).toHaveLength(3); + await act(async () => editButtons[1]?.click()); + expect(container.querySelector("textarea")?.value).toBe("then second."); + expect(button("Done editing").disabled).toBe(true); + await click("Save"); + expect(mocks.edit).toHaveBeenCalledWith("sentence-test", 2, "then second."); + await click("Done editing"); + expect(button("00:00Speaker AFirst, then second.")).toBeDefined(); + }); + + it("groups translated and provisional live captions", async () => { + mocks.language = "ru"; + await act(async () => root.render(createElement(Transcript, { data }))); + expect(button("00:00Speaker AFirst, then second.")).toBeDefined(); + expect(container.textContent).not.toContain("Edit transcript"); + mocks.live = true; + await act(async () => + root.render( + createElement(Transcript, { + data: { ...data, transcriptionStatus: "PROCESSING" }, + }), + ), + ); + expect(container.textContent).toContain("Live transcript"); + expect(button("00:00Speaker AFirst, then second.")).toBeDefined(); + }); +}); diff --git a/apps/web/__tests__/unit/transcript-sentences.test.ts b/apps/web/__tests__/unit/transcript-sentences.test.ts new file mode 100644 index 00000000000..0bf4938bf47 --- /dev/null +++ b/apps/web/__tests__/unit/transcript-sentences.test.ts @@ -0,0 +1,119 @@ +import { describe, expect, it } from "vitest"; +import { formatToWebVTT } from "@/lib/transcribe-utils"; +import { groupTranscriptSentences } from "@/lib/transcript-sentences"; +import { parseVTT, type TranscriptEntry } from "@/lib/transcript-vtt"; + +function entries(texts: string[], speaker: string | null = "A") { + return texts.map( + (text, index): TranscriptEntry => ({ + id: index + 1, + text, + startTime: index * 2 + 0.125, + endTime: index * 2 + 1, + timestamp: `00:${String(index * 2).padStart(2, "0")}`, + speaker, + }), + ); +} + +describe("groupTranscriptSentences", () => { + it("joins comma, pause and eight-word caption breaks into a full sentence", () => { + const words = + "Immediate release, extended release, это всё одна фраза которую нужно читать целиком." + .split(" ") + .map((text, index) => ({ + text, + start: index * 1_000, + end: index * 1_000 + 300, + speaker: "A", + })); + const captions = parseVTT(formatToWebVTT({ words })); + expect(captions.length).toBeGreaterThan(8); + expect(groupTranscriptSentences(captions)).toEqual([ + { + ...captions[0], + text: words.map((word) => word.text).join(" "), + endTime: 11.3, + }, + ]); + }); + + it("uses sentence punctuation, including closing quotes and CJK punctuation", () => { + const result = groupTranscriptSentences( + entries([ + "И как бы,", + "сейчас,", + "где она здесь?", + "Он сказал:", + "«Вот это.»", + "真的。", + "Next!", + ]), + ); + expect(result.map((entry) => entry.text)).toEqual([ + "И как бы, сейчас, где она здесь?", + "Он сказал: «Вот это.»", + "真的。", + "Next!", + ]); + }); + + it("keeps abbreviations and hesitation ellipses with the following phrase", () => { + expect( + groupTranscriptSentences( + entries(["Dr.", "Smith said,", "I want...", "to continue."]), + ).map((entry) => entry.text), + ).toEqual(["Dr. Smith said, I want... to continue."]); + }); + + it("never combines different or unknown speakers", () => { + const source = entries(["First,", "second,", "unknown,", "fourth."]); + if (source[1]) source[1].speaker = "B"; + if (source[2]) source[2].speaker = null; + expect(groupTranscriptSentences(source)).toEqual(source); + }); + + it("keeps the first cue identity and exact outer timestamps without mutating cues", () => { + const source = entries(["One,", "two."]); + const original = structuredClone(source); + expect(groupTranscriptSentences(source)[0]).toEqual({ + ...source[0], + text: "One, two.", + endTime: 3, + }); + expect(source).toEqual(original); + }); + + it("breaks at long silence and overlapping cues", () => { + const source = entries(["Before", "after"]); + if (source[1]) source[1].startTime = 6; + expect(groupTranscriptSentences(source)).toHaveLength(2); + if (source[1]) source[1].startTime = 0.5; + expect(groupTranscriptSentences(source)).toHaveLength(2); + }); + + it("bounds unpunctuated speech and retains every fragment", () => { + const source = entries( + Array.from({ length: 100 }, () => "a long unpunctuated fragment"), + ); + const grouped = groupTranscriptSentences(source); + expect(grouped.length).toBeGreaterThan(1); + expect( + grouped.every( + (entry) => + entry.text.length <= 800 && entry.endTime - entry.startTime <= 45, + ), + ).toBe(true); + expect(grouped.map((entry) => entry.text).join(" ")).toBe( + source.map((entry) => entry.text).join(" "), + ); + }); + + it("handles empty input and unfinished live sentences", () => { + expect(groupTranscriptSentences([])).toEqual([]); + expect(groupTranscriptSentences(entries([" "]))).toEqual([]); + expect( + groupTranscriptSentences(entries(["still", "speaking"]))[0]?.text, + ).toBe("still speaking"); + }); +}); diff --git a/apps/web/app/s/[videoId]/_components/tabs/Transcript.tsx b/apps/web/app/s/[videoId]/_components/tabs/Transcript.tsx index 4f6a9557773..c4d65b69333 100644 --- a/apps/web/app/s/[videoId]/_components/tabs/Transcript.tsx +++ b/apps/web/app/s/[videoId]/_components/tabs/Transcript.tsx @@ -21,6 +21,7 @@ import { SUPPORTED_LANGUAGES, } from "@/actions/videos/translation-languages"; import { useCurrentUser } from "@/app/Layout/AuthContext"; +import { groupTranscriptSentences } from "@/lib/transcript-sentences"; import { formatTranscriptAsParagraphs } from "@/lib/transcript-text"; import { formatVttCueText, @@ -51,6 +52,7 @@ export const Transcript: React.FC = ({ data, onSeek }) => { const [transcriptData, setTranscriptData] = useState([]); const [selectedEntry, setSelectedEntry] = useState(null); const [retryTriggered, setRetryTriggered] = useState(false); + const [isEditingTranscript, setIsEditingTranscript] = useState(false); const [editingEntry, setEditingEntry] = useState(null); const [editText, setEditText] = useState(""); const [isSaving, setIsSaving] = useState(false); @@ -163,7 +165,10 @@ export const Transcript: React.FC = ({ data, onSeek }) => { const isLiveTranscriptActive = liveTranscript?.kind === "ready" && liveTranscript.state === "active"; const liveTranscriptData = useMemo( - () => (liveTranscriptContent ? parseVTT(liveTranscriptContent) : []), + () => + liveTranscriptContent + ? groupTranscriptSentences(parseVTT(liveTranscriptContent)) + : [], [liveTranscriptContent], ); @@ -216,6 +221,11 @@ export const Transcript: React.FC = ({ data, onSeek }) => { } }, [captionContext.currentVttContent, transcriptContent, selectedLanguage]); + const sentenceData = useMemo( + () => groupTranscriptSentences(transcriptData), + [transcriptData], + ); + const handleLanguageChange = async (language: CaptionLanguage) => { setShowLanguageMenu(false); captionContext.setSelectedLanguage(language); @@ -384,7 +394,7 @@ export const Transcript: React.FC = ({ data, onSeek }) => { const copyTimestampedTranscript = () => { if (transcriptData.length === 0) return; - void copyTranscriptText(formatTranscriptForClipboard(transcriptData)); + void copyTranscriptText(formatTranscriptForClipboard(sentenceData)); }; const triggerTranscriptDownload = ( @@ -429,6 +439,8 @@ export const Transcript: React.FC = ({ data, onSeek }) => { }; const canEdit = user?.id === data.owner.id && selectedLanguage === "original"; + const showEditingControls = canEdit && isEditingTranscript; + const displayedEntries = showEditingControls ? transcriptData : sentenceData; const liveTranscriptView = showLiveTranscript ? (
@@ -634,7 +646,11 @@ export const Transcript: React.FC = ({ data, onSeek }) => {
+ {canEdit && ( + + )}
@@ -807,7 +834,7 @@ export const Transcript: React.FC = ({ data, onSeek }) => {
)}
- {transcriptData.map((entry) => ( + {displayedEntries.map((entry) => (
= ({ data, onSeek }) => { {entry.text} - {canEdit && ( + {showEditingControls && (