diff --git a/frontend/package-lock.json b/frontend/package-lock.json index 5a2dade..5a02b7a 100644 --- a/frontend/package-lock.json +++ b/frontend/package-lock.json @@ -23,6 +23,8 @@ "jszip": "^3.10.1", "lucide-react": "^0.462.0", "mermaid": "^11.16.0", + "prosemirror-model": "^1.25.9", + "prosemirror-state": "^1.4.4", "react": "^18.3.1", "react-dom": "^18.3.1", "react-markdown": "^10.1.0", diff --git a/frontend/package.json b/frontend/package.json index 7d0e65a..5add789 100644 --- a/frontend/package.json +++ b/frontend/package.json @@ -26,6 +26,8 @@ "jszip": "^3.10.1", "lucide-react": "^0.462.0", "mermaid": "^11.16.0", + "prosemirror-model": "^1.25.9", + "prosemirror-state": "^1.4.4", "react": "^18.3.1", "react-dom": "^18.3.1", "react-markdown": "^10.1.0", diff --git a/frontend/src/utils/dictationInsert.test.ts b/frontend/src/utils/dictationInsert.test.ts new file mode 100644 index 0000000..4c4ea2c --- /dev/null +++ b/frontend/src/utils/dictationInsert.test.ts @@ -0,0 +1,148 @@ +import { Schema, type Node as PMNode } from 'prosemirror-model' +import { EditorState, NodeSelection, TextSelection } from 'prosemirror-state' +import { describe, expect, it } from 'vitest' + +import { insertDictationAtSelection, padDictatedText } from '@/utils/dictationInsert' + +const schema = new Schema({ + nodes: { + doc: { content: 'block+' }, + paragraph: { group: 'block', content: 'inline*' }, + image: { group: 'block', atom: true }, + text: { group: 'inline' }, + hard_break: { group: 'inline', inline: true, leafText: () => '\n' }, + mention: { group: 'inline', inline: true, atom: true }, + }, + marks: { strong: {} }, +}) + +const p = (...content: (string | PMNode)[]) => + schema.node('paragraph', null, content.map((c) => (typeof c === 'string' ? schema.text(c) : c))) +const bold = (text: string) => schema.text(text, [schema.mark('strong')]) + +// Builds a one-paragraph-per-arg doc and puts a text cursor at `|` (or a +// range between two `|`s) in the first paragraph that has one. +function stateWithCursor(...paragraphs: string[]) { + let anchor = -1 + let head = -1 + let pos = 0 + const nodes = paragraphs.map((src) => { + const parts = src.split('|') + let text = '' + parts.forEach((part, i) => { + if (i > 0) { + const at = pos + 1 + text.length + if (anchor < 0) anchor = at + else head = at + } + text += part + }) + const node = text ? p(text) : p() + pos += node.nodeSize + return node + }) + const doc = schema.node('doc', null, nodes) + return EditorState.create({ doc, selection: TextSelection.create(doc, anchor, head < 0 ? anchor : head) }) +} + +// Dictates each chunk in turn, the way consecutive speech results arrive. +function dictate(state: EditorState, ...chunks: string[]) { + for (const chunk of chunks) { + const tr = state.tr + expect(insertDictationAtSelection(tr, chunk)).toBe(true) + state = state.apply(tr) + } + return state +} + +// Paragraph texts with the caret marked as `|`. +function show(state: EditorState) { + const { from } = state.selection + const out: string[] = [] + state.doc.forEach((node, offset) => { + const start = offset + 1 + const text = node.textBetween(0, node.content.size, undefined, (n) => n.type.spec.leafText?.(n) ?? '@') + out.push(from >= start && from <= start + node.content.size + ? text.slice(0, from - start) + '|' + text.slice(from - start) + : text) + }) + return out +} + +describe('insertDictationAtSelection', () => { + it('types at the cursor in the middle of a note, not at the bottom', () => { + const state = dictate(stateWithCursor('First line', 'Hello |world', 'Last line'), 'there') + expect(show(state)).toEqual(['First line', 'Hello there |world', 'Last line']) + }) + + it('continues each new chunk from where the last one ended, in order', () => { + const state = dictate(stateWithCursor('Hello |world'), 'one', 'two', 'three') + expect(show(state)).toEqual(['Hello one two three |world']) + }) + + it('adds a leading space after a word and none into an empty line', () => { + expect(show(dictate(stateWithCursor('Hello|'), 'there'))).toEqual(['Hello there|']) + expect(show(dictate(stateWithCursor('|'), 'Hello'))).toEqual(['Hello|']) + }) + + it('does not space away from surrounding punctuation or brackets', () => { + expect(show(dictate(stateWithCursor('Hello|.'), 'there'))).toEqual(['Hello there|.']) + expect(show(dictate(stateWithCursor('(|)'), 'aside'))).toEqual(['(aside|)']) + }) + + it('collapses a range selection to its end instead of overwriting it', () => { + const state = dictate(stateWithCursor('keep |this| text'), 'and more') + expect(show(state)).toEqual(['keep this and more| text']) + expect(state.selection.empty).toBe(true) + }) + + it('normalizes whitespace and newlines in the recognized text', () => { + expect(show(dictate(stateWithCursor('|'), ' one\n two '))).toEqual(['one two|']) + }) + + it('picks up the marks at the cursor, like typing does', () => { + const doc = schema.node('doc', null, [p(bold('bold'))]) + const state = dictate(EditorState.create({ doc, selection: TextSelection.create(doc, 3) }), 'x') + const para = state.doc.firstChild! + expect(para.textContent).toBe('bo x ld') + expect(para.childCount).toBe(1) + expect(para.firstChild!.marks.map((m) => m.type.name)).toEqual(['strong']) + }) + + it('spaces away from an inline chip but not from a line break', () => { + const afterBreak = schema.node('doc', null, [p('a', schema.node('hard_break'))]) + const s1 = dictate(EditorState.create({ doc: afterBreak, selection: TextSelection.atEnd(afterBreak) }), 'b') + expect(show(s1)).toEqual(['a\nb|']) + + const afterChip = schema.node('doc', null, [p('see ', schema.node('mention'))]) + const s2 = dictate(EditorState.create({ doc: afterChip, selection: TextSelection.atEnd(afterChip) }), 'this') + expect(show(s2)).toEqual(['see @ this|']) + }) + + it('refuses (leaving the transaction untouched) when the selection cannot hold text', () => { + const doc = schema.node('doc', null, [p('text'), schema.node('image')]) + const state = EditorState.create({ doc, selection: NodeSelection.create(doc, 6) }) + const tr = state.tr + expect(insertDictationAtSelection(tr, 'hello')).toBe(false) + expect(tr.docChanged).toBe(false) + expect(tr.selectionSet).toBe(false) + }) + + it('ignores empty or whitespace-only results', () => { + const tr = stateWithCursor('Hello|').tr + expect(insertDictationAtSelection(tr, ' ')).toBe(false) + expect(tr.docChanged).toBe(false) + }) +}) + +describe('padDictatedText', () => { + it('pads only where the neighbours are word characters', () => { + expect(padDictatedText('b', 'a', 'c')).toBe(' b ') + expect(padDictatedText('b', '', '')).toBe('b') + expect(padDictatedText('b', ' ', ' ')).toBe('b') + }) + + it('does not put a space before text that starts with punctuation', () => { + expect(padDictatedText(', and then', 'a', '')).toBe(', and then') + }) +}) diff --git a/frontend/src/utils/dictationInsert.ts b/frontend/src/utils/dictationInsert.ts new file mode 100644 index 0000000..92f31cc --- /dev/null +++ b/frontend/src/utils/dictationInsert.ts @@ -0,0 +1,61 @@ +import type { Node as PMNode } from 'prosemirror-model' +import { TextSelection, type Transaction } from 'prosemirror-state' + +// Characters that a dictated phrase shouldn't be separated from by a space: +// nothing is needed after whitespace or an opening bracket/quote, and before +// whitespace or closing punctuation. +const NO_SPACE_AFTER = /[\s([{‘“]/ +const NO_SPACE_BEFORE = /[\s.,;:!?)\]}’”]/ + +/** Speech results can carry stray leading/trailing spaces or line breaks; a + * dictated chunk is always inserted as a single run of words. */ +export function normalizeDictatedText(text: string): string { + return text.trim().replace(/\s+/g, ' ') +} + +/** + * Pads dictated `text` with spaces so it doesn't fuse onto the characters + * immediately `before` and `after` the insertion point (empty string when the + * insertion point is at the start/end of its line). + */ +export function padDictatedText(text: string, before: string, after: string): string { + let padded = text + if (before && !NO_SPACE_AFTER.test(before) && !NO_SPACE_BEFORE.test(text[0])) padded = ` ${padded}` + if (after && !NO_SPACE_BEFORE.test(after)) padded = `${padded} ` + return padded +} + +// Inline nodes without text (a hard break, a mention chip) still count as +// something to space away from; ones that render as whitespace don't. +const leafText = (node: PMNode) => node.type.spec.leafText?.(node) ?? '' + +/** + * Types `text` into the document at the transaction's selection, the way the + * keyboard would: a range selection is collapsed to its end first (dictating + * never overwrites what's selected), the words pick up the marks at that + * position, and the caret is left right after them so the next dictated + * chunk continues from there. + * + * ProseMirror keeps its selection when the editor loses focus, so this works + * even while focus sits on the mic button rather than in the note. + * + * Returns false without touching `tr` when the selection isn't in a text + * block (e.g. an image block is selected, or the whole document), so the + * caller can fall back to starting a new paragraph. + */ +export function insertDictationAtSelection(tr: Transaction, text: string): boolean { + const words = normalizeDictatedText(text) + const $pos = tr.selection.$to + const parent = $pos.parent + if (!words || !parent.inlineContent) return false + + const offset = $pos.parentOffset + const before = parent.textBetween(Math.max(0, offset - 1), offset, undefined, leafText) + const after = parent.textBetween(offset, Math.min(parent.content.size, offset + 1), undefined, leafText) + const padded = padDictatedText(words, before, after) + + tr.insertText(padded, $pos.pos) + tr.setSelection(TextSelection.create(tr.doc, $pos.pos + padded.length)) + tr.scrollIntoView() + return true +} diff --git a/frontend/src/views/EditorView.tsx b/frontend/src/views/EditorView.tsx index cd370a9..f04d687 100644 --- a/frontend/src/views/EditorView.tsx +++ b/frontend/src/views/EditorView.tsx @@ -49,7 +49,8 @@ import { transcriptionApi } from '@/api/transcription' import { notesApi, configApi, type Note } from '@/api/notes' import { foldersApi, type Folder } from '@/api/folders' import { annotationsApi, type Annotation } from '@/api/annotations' -import { useDictation, type DictationMode } from '@/hooks/useDictation' +import { useDictation } from '@/hooks/useDictation' +import { insertDictationAtSelection, normalizeDictatedText } from '@/utils/dictationInsert' import { useTextToSpeech } from '@/hooks/useTextToSpeech' import { extractPlainText } from '@/utils/blocks' import { noteToMarkdownBody, svgToPngData } from '@/utils/export' @@ -259,16 +260,10 @@ export default function EditorView() { }, }) - // Insert blocks after `anchorBlockId` when given and still present (used by - // dictation — see dictationAnchorBlockIdRef below), otherwise at the cursor - // when the editor is focused, otherwise append to the end of the note. - // Returns the inserted blocks so callers (e.g. dictation) can track them. - const insertBlocksAtCursor = useCallback((blocks: PartialBlock[], anchorBlockId?: string | null) => { + // Insert blocks at the cursor when the editor is focused, otherwise append to + // the end of the note. Returns the inserted blocks. + const insertBlocksAtCursor = useCallback((blocks: PartialBlock[]) => { if (!editor || blocks.length === 0) return [] - if (anchorBlockId) { - const anchor = editor.getBlock(anchorBlockId) - if (anchor) return editor.insertBlocks(blocks, anchor, 'after') - } if (editor.isFocused()) { const cursorBlock = editor.getTextCursorPosition().block return editor.insertBlocks(blocks, cursorBlock, 'after') @@ -302,50 +297,42 @@ export default function EditorView() { if (firstBlock) editor.insertBlocks(blocks, firstBlock, 'before') }, [editor]) - // Tracks the paragraph block the *current* dictation session is appending - // to, so consecutive recognized chunks concatenate onto one line instead of - // each becoming its own new block (which, without a stable insertion point, - // ends up stacking in reverse order as the cursor never advances). Cleared - // whenever a dictation session isn't active — see the effect below. - const dictationModeRef = useRef(null) - const dictationSessionBlockIdRef = useRef(null) - // The block the cursor was in when the *current* dictation session started - // — captured before dictation moves focus to the mic button (so the note - // editor is no longer "focused" for the rest of the session). Without this, - // every dictated chunk would fall back to appending at the end of the note - // instead of landing where the user had their cursor. - const dictationAnchorBlockIdRef = useRef(null) - + // Whether the user has put the text cursor somewhere in this note's body. + // ProseMirror keeps its selection when the editor blurs, so once they have, + // dictation can keep typing there even though focus moves to the mic button + // (on click, and deliberately once a session starts, so Enter/Space toggle + // it). Set by the editor wrapper's onFocus; reset whenever a note is loaded + // into the editor — see the hydrate effect. + const editorHasCursorRef = useRef(false) + + // Dictated text goes where the note's cursor is, like typing, and each chunk + // continues from where the last one ended. With no cursor placed in the note + // it starts a line at the bottom instead, and leaves the cursor there so the + // rest of the session carries on along that line. const insertDictatedText = useCallback((text: string) => { - const trimmed = text.trim() - if (!trimmed || !editor) return - - const inSession = dictationModeRef.current === 'dictation' - const targetId = inSession ? dictationSessionBlockIdRef.current : null - - if (targetId) { - const existing = editor.getBlock(targetId) - if (existing) { - const priorText = extractPlainText([existing]) - const merged = priorText ? `${priorText} ${trimmed}` : trimmed - editor.updateBlock(targetId, { content: [{ type: 'text', text: merged, styles: {} }] }) - if (editor.isFocused()) editor.setTextCursorPosition(targetId, 'end') - return - } - // Target block was deleted mid-session (e.g. user backspaced it) — fall - // through and re-anchor to a freshly inserted one. - } + const words = normalizeDictatedText(text) + if (!words || !editor) return - const inserted = insertBlocksAtCursor( - [{ type: 'paragraph', content: [{ type: 'text', text: trimmed, styles: {} }] }], - inSession ? dictationAnchorBlockIdRef.current : null, - ) - const newBlock = inserted[0] - if (newBlock) { - if (inSession) dictationSessionBlockIdRef.current = newBlock.id - if (editor.isFocused()) editor.setTextCursorPosition(newBlock.id, 'end') + const hasCursor = editorHasCursorRef.current || editor.isFocused() + if (hasCursor && editor.transact((tr) => insertDictationAtSelection(tr, words))) return + + // No cursor, or it's on a block that can't hold text (e.g. an image): + // start a new paragraph after that block, or at the bottom of the note — + // filling the note's trailing empty line rather than leaving a gap above. + const doc = editor.document + const anchor = hasCursor ? editor.getTextCursorPosition().block : doc[doc.length - 1] + if (!anchor) return + let targetId = anchor.id + if (!hasCursor && anchor.type === 'paragraph' && anchor.content.length === 0 && anchor.children.length === 0) { + editor.updateBlock(anchor, { content: words }) + } else { + const [inserted] = editor.insertBlocks([{ type: 'paragraph', content: words }], anchor, 'after') + if (!inserted) return + targetId = inserted.id } - }, [editor, insertBlocksAtCursor]) + editor.setTextCursorPosition(targetId, 'end') + editorHasCursorRef.current = true + }, [editor]) // Upload an audio blob to /media and return its URL. The filename extension // must match the blob type so the backend accepts it (.webm/.ogg/.mp3 are allowed). @@ -386,32 +373,6 @@ export default function EditorView() { sttProvider, }) - // Clear the dictation session's target block whenever a session isn't - // active, so the next session starts a fresh paragraph rather than - // continuing to append to a stale one. - useEffect(() => { - dictationModeRef.current = dictation.mode - if (dictation.mode !== 'dictation') { - dictationSessionBlockIdRef.current = null - dictationAnchorBlockIdRef.current = null - } - }, [dictation.mode]) - - // Dictation now focuses the mic button as soon as it starts (so Enter/Space - // can toggle it), which means the editor is no longer "focused" by the time - // insertDictatedText runs. So capture the cursor's block *here*, synchronously, - // right before that focus change happens — this is the last moment the editor - // still reports itself focused for a click that's about to start dictation. - const handleDictationToggle = useCallback(() => { - const willStart = dictation.status === 'idle' || dictation.status === 'error' - if (willStart) { - dictationAnchorBlockIdRef.current = editor?.isFocused() - ? editor.getTextCursorPosition().block.id - : null - } - dictation.toggleDictation() - }, [dictation, editor]) - // Upload a recorded video blob to /media and return its URL + stored filename // (the filename is what the async transcription job references). const uploadVideoBlob = useCallback(async (blob: Blob, mimeType: string, baseName: string): Promise<{ url: string; filename: string }> => { @@ -740,6 +701,9 @@ export default function EditorView() { const blocks = isNewNote ? EMPTY_DOCUMENT : parseNoteContent(note?.content ?? '[]') isHydratingEditor.current = true editor.replaceBlocks(editor.document, blocks as Parameters[1]) + // The selection just landed wherever the replaced content left it, not + // somewhere the user chose — unless they're in the editor right now. + editorHasCursorRef.current = editor.isFocused() currentNoteContent.current = extractPlainText(blocks as unknown[]) syncedEditorKey.current = editorKey hasPendingChanges.current = false @@ -1852,7 +1816,7 @@ export default function EditorView() { anchorRef={exportAnchorRef} onPlayPause={handlePlayPause} dictation={dictation} - onDictationToggle={handleDictationToggle} + onDictationToggle={dictation.toggleDictation} onRecordToggle={dictation.toggleRecording} insertMode={ttsInsertMode} onToggleInsertMode={() => setTtsInsertMode((v) => !v)} @@ -1893,7 +1857,13 @@ export default function EditorView() { -
+
{ if (editor.isFocused()) editorHasCursorRef.current = true }} + > setTtsInsertMode((v) => !v)}