From b62b8e45fd3c42d76206aea30df09385f368530d Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 26 Sep 2026 18:38:20 +0000 Subject: [PATCH] Type dictated words at the note's cursor instead of the bottom Clicking the mic moves focus off the editor before its click handler runs, so the "is the editor focused?" check used to capture the cursor always failed and every dictated chunk was appended to the end of the note. Even when it did catch the cursor, it started a new paragraph after the cursor's block rather than typing at the cursor. ProseMirror keeps its selection when the editor blurs, so dictation now types straight into that selection (collapsing a range to its end, with spacing to neighbouring words), leaving the caret after the words so each chunk continues in order. This works however focus moves, so the mic button keeps focus for Enter/Space toggling, and it follows the cursor if the user clicks elsewhere mid-session. With no cursor placed in the note (tracked via the editor wrapper's focus, reset when a note is loaded) the words go at the bottom as before, now filling the trailing empty line instead of leaving a gap. prosemirror-state/-model are declared as direct dependencies at the versions already installed through BlockNote. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_017Go397yGvJqhvMQhc53Q46 --- frontend/package-lock.json | 2 + frontend/package.json | 2 + frontend/src/utils/dictationInsert.test.ts | 148 +++++++++++++++++++++ frontend/src/utils/dictationInsert.ts | 61 +++++++++ frontend/src/views/EditorView.tsx | 130 +++++++----------- 5 files changed, 263 insertions(+), 80 deletions(-) create mode 100644 frontend/src/utils/dictationInsert.test.ts create mode 100644 frontend/src/utils/dictationInsert.ts diff --git a/frontend/package-lock.json b/frontend/package-lock.json index 5a2dade..5a02b7a 100644 --- a/frontend/package-lock.json +++ b/frontend/package-lock.json @@ -23,6 +23,8 @@ "jszip": "^3.10.1", "lucide-react": "^0.462.0", "mermaid": "^11.16.0", + "prosemirror-model": "^1.25.9", + "prosemirror-state": "^1.4.4", "react": "^18.3.1", "react-dom": "^18.3.1", "react-markdown": "^10.1.0", diff --git a/frontend/package.json b/frontend/package.json index 7d0e65a..5add789 100644 --- a/frontend/package.json +++ b/frontend/package.json @@ -26,6 +26,8 @@ "jszip": "^3.10.1", "lucide-react": "^0.462.0", "mermaid": "^11.16.0", + "prosemirror-model": "^1.25.9", + "prosemirror-state": "^1.4.4", "react": "^18.3.1", "react-dom": "^18.3.1", "react-markdown": "^10.1.0", diff --git a/frontend/src/utils/dictationInsert.test.ts b/frontend/src/utils/dictationInsert.test.ts new file mode 100644 index 0000000..4c4ea2c --- /dev/null +++ b/frontend/src/utils/dictationInsert.test.ts @@ -0,0 +1,148 @@ +import { Schema, type Node as PMNode } from 'prosemirror-model' +import { EditorState, NodeSelection, TextSelection } from 'prosemirror-state' +import { describe, expect, it } from 'vitest' + +import { insertDictationAtSelection, padDictatedText } from '@/utils/dictationInsert' + +const schema = new Schema({ + nodes: { + doc: { content: 'block+' }, + paragraph: { group: 'block', content: 'inline*' }, + image: { group: 'block', atom: true }, + text: { group: 'inline' }, + hard_break: { group: 'inline', inline: true, leafText: () => '\n' }, + mention: { group: 'inline', inline: true, atom: true }, + }, + marks: { strong: {} }, +}) + +const p = (...content: (string | PMNode)[]) => + schema.node('paragraph', null, content.map((c) => (typeof c === 'string' ? schema.text(c) : c))) +const bold = (text: string) => schema.text(text, [schema.mark('strong')]) + +// Builds a one-paragraph-per-arg doc and puts a text cursor at `|` (or a +// range between two `|`s) in the first paragraph that has one. +function stateWithCursor(...paragraphs: string[]) { + let anchor = -1 + let head = -1 + let pos = 0 + const nodes = paragraphs.map((src) => { + const parts = src.split('|') + let text = '' + parts.forEach((part, i) => { + if (i > 0) { + const at = pos + 1 + text.length + if (anchor < 0) anchor = at + else head = at + } + text += part + }) + const node = text ? p(text) : p() + pos += node.nodeSize + return node + }) + const doc = schema.node('doc', null, nodes) + return EditorState.create({ doc, selection: TextSelection.create(doc, anchor, head < 0 ? anchor : head) }) +} + +// Dictates each chunk in turn, the way consecutive speech results arrive. +function dictate(state: EditorState, ...chunks: string[]) { + for (const chunk of chunks) { + const tr = state.tr + expect(insertDictationAtSelection(tr, chunk)).toBe(true) + state = state.apply(tr) + } + return state +} + +// Paragraph texts with the caret marked as `|`. +function show(state: EditorState) { + const { from } = state.selection + const out: string[] = [] + state.doc.forEach((node, offset) => { + const start = offset + 1 + const text = node.textBetween(0, node.content.size, undefined, (n) => n.type.spec.leafText?.(n) ?? '@') + out.push(from >= start && from <= start + node.content.size + ? text.slice(0, from - start) + '|' + text.slice(from - start) + : text) + }) + return out +} + +describe('insertDictationAtSelection', () => { + it('types at the cursor in the middle of a note, not at the bottom', () => { + const state = dictate(stateWithCursor('First line', 'Hello |world', 'Last line'), 'there') + expect(show(state)).toEqual(['First line', 'Hello there |world', 'Last line']) + }) + + it('continues each new chunk from where the last one ended, in order', () => { + const state = dictate(stateWithCursor('Hello |world'), 'one', 'two', 'three') + expect(show(state)).toEqual(['Hello one two three |world']) + }) + + it('adds a leading space after a word and none into an empty line', () => { + expect(show(dictate(stateWithCursor('Hello|'), 'there'))).toEqual(['Hello there|']) + expect(show(dictate(stateWithCursor('|'), 'Hello'))).toEqual(['Hello|']) + }) + + it('does not space away from surrounding punctuation or brackets', () => { + expect(show(dictate(stateWithCursor('Hello|.'), 'there'))).toEqual(['Hello there|.']) + expect(show(dictate(stateWithCursor('(|)'), 'aside'))).toEqual(['(aside|)']) + }) + + it('collapses a range selection to its end instead of overwriting it', () => { + const state = dictate(stateWithCursor('keep |this| text'), 'and more') + expect(show(state)).toEqual(['keep this and more| text']) + expect(state.selection.empty).toBe(true) + }) + + it('normalizes whitespace and newlines in the recognized text', () => { + expect(show(dictate(stateWithCursor('|'), ' one\n two '))).toEqual(['one two|']) + }) + + it('picks up the marks at the cursor, like typing does', () => { + const doc = schema.node('doc', null, [p(bold('bold'))]) + const state = dictate(EditorState.create({ doc, selection: TextSelection.create(doc, 3) }), 'x') + const para = state.doc.firstChild! + expect(para.textContent).toBe('bo x ld') + expect(para.childCount).toBe(1) + expect(para.firstChild!.marks.map((m) => m.type.name)).toEqual(['strong']) + }) + + it('spaces away from an inline chip but not from a line break', () => { + const afterBreak = schema.node('doc', null, [p('a', schema.node('hard_break'))]) + const s1 = dictate(EditorState.create({ doc: afterBreak, selection: TextSelection.atEnd(afterBreak) }), 'b') + expect(show(s1)).toEqual(['a\nb|']) + + const afterChip = schema.node('doc', null, [p('see ', schema.node('mention'))]) + const s2 = dictate(EditorState.create({ doc: afterChip, selection: TextSelection.atEnd(afterChip) }), 'this') + expect(show(s2)).toEqual(['see @ this|']) + }) + + it('refuses (leaving the transaction untouched) when the selection cannot hold text', () => { + const doc = schema.node('doc', null, [p('text'), schema.node('image')]) + const state = EditorState.create({ doc, selection: NodeSelection.create(doc, 6) }) + const tr = state.tr + expect(insertDictationAtSelection(tr, 'hello')).toBe(false) + expect(tr.docChanged).toBe(false) + expect(tr.selectionSet).toBe(false) + }) + + it('ignores empty or whitespace-only results', () => { + const tr = stateWithCursor('Hello|').tr + expect(insertDictationAtSelection(tr, ' ')).toBe(false) + expect(tr.docChanged).toBe(false) + }) +}) + +describe('padDictatedText', () => { + it('pads only where the neighbours are word characters', () => { + expect(padDictatedText('b', 'a', 'c')).toBe(' b ') + expect(padDictatedText('b', '', '')).toBe('b') + expect(padDictatedText('b', ' ', ' ')).toBe('b') + }) + + it('does not put a space before text that starts with punctuation', () => { + expect(padDictatedText(', and then', 'a', '')).toBe(', and then') + }) +}) diff --git a/frontend/src/utils/dictationInsert.ts b/frontend/src/utils/dictationInsert.ts new file mode 100644 index 0000000..92f31cc --- /dev/null +++ b/frontend/src/utils/dictationInsert.ts @@ -0,0 +1,61 @@ +import type { Node as PMNode } from 'prosemirror-model' +import { TextSelection, type Transaction } from 'prosemirror-state' + +// Characters that a dictated phrase shouldn't be separated from by a space: +// nothing is needed after whitespace or an opening bracket/quote, and before +// whitespace or closing punctuation. +const NO_SPACE_AFTER = /[\s([{‘“]/ +const NO_SPACE_BEFORE = /[\s.,;:!?)\]}’”]/ + +/** Speech results can carry stray leading/trailing spaces or line breaks; a + * dictated chunk is always inserted as a single run of words. */ +export function normalizeDictatedText(text: string): string { + return text.trim().replace(/\s+/g, ' ') +} + +/** + * Pads dictated `text` with spaces so it doesn't fuse onto the characters + * immediately `before` and `after` the insertion point (empty string when the + * insertion point is at the start/end of its line). + */ +export function padDictatedText(text: string, before: string, after: string): string { + let padded = text + if (before && !NO_SPACE_AFTER.test(before) && !NO_SPACE_BEFORE.test(text[0])) padded = ` ${padded}` + if (after && !NO_SPACE_BEFORE.test(after)) padded = `${padded} ` + return padded +} + +// Inline nodes without text (a hard break, a mention chip) still count as +// something to space away from; ones that render as whitespace don't. +const leafText = (node: PMNode) => node.type.spec.leafText?.(node) ?? '' + +/** + * Types `text` into the document at the transaction's selection, the way the + * keyboard would: a range selection is collapsed to its end first (dictating + * never overwrites what's selected), the words pick up the marks at that + * position, and the caret is left right after them so the next dictated + * chunk continues from there. + * + * ProseMirror keeps its selection when the editor loses focus, so this works + * even while focus sits on the mic button rather than in the note. + * + * Returns false without touching `tr` when the selection isn't in a text + * block (e.g. an image block is selected, or the whole document), so the + * caller can fall back to starting a new paragraph. + */ +export function insertDictationAtSelection(tr: Transaction, text: string): boolean { + const words = normalizeDictatedText(text) + const $pos = tr.selection.$to + const parent = $pos.parent + if (!words || !parent.inlineContent) return false + + const offset = $pos.parentOffset + const before = parent.textBetween(Math.max(0, offset - 1), offset, undefined, leafText) + const after = parent.textBetween(offset, Math.min(parent.content.size, offset + 1), undefined, leafText) + const padded = padDictatedText(words, before, after) + + tr.insertText(padded, $pos.pos) + tr.setSelection(TextSelection.create(tr.doc, $pos.pos + padded.length)) + tr.scrollIntoView() + return true +} diff --git a/frontend/src/views/EditorView.tsx b/frontend/src/views/EditorView.tsx index cd370a9..f04d687 100644 --- a/frontend/src/views/EditorView.tsx +++ b/frontend/src/views/EditorView.tsx @@ -49,7 +49,8 @@ import { transcriptionApi } from '@/api/transcription' import { notesApi, configApi, type Note } from '@/api/notes' import { foldersApi, type Folder } from '@/api/folders' import { annotationsApi, type Annotation } from '@/api/annotations' -import { useDictation, type DictationMode } from '@/hooks/useDictation' +import { useDictation } from '@/hooks/useDictation' +import { insertDictationAtSelection, normalizeDictatedText } from '@/utils/dictationInsert' import { useTextToSpeech } from '@/hooks/useTextToSpeech' import { extractPlainText } from '@/utils/blocks' import { noteToMarkdownBody, svgToPngData } from '@/utils/export' @@ -259,16 +260,10 @@ export default function EditorView() { }, }) - // Insert blocks after `anchorBlockId` when given and still present (used by - // dictation — see dictationAnchorBlockIdRef below), otherwise at the cursor - // when the editor is focused, otherwise append to the end of the note. - // Returns the inserted blocks so callers (e.g. dictation) can track them. - const insertBlocksAtCursor = useCallback((blocks: PartialBlock[], anchorBlockId?: string | null) => { + // Insert blocks at the cursor when the editor is focused, otherwise append to + // the end of the note. Returns the inserted blocks. + const insertBlocksAtCursor = useCallback((blocks: PartialBlock[]) => { if (!editor || blocks.length === 0) return [] - if (anchorBlockId) { - const anchor = editor.getBlock(anchorBlockId) - if (anchor) return editor.insertBlocks(blocks, anchor, 'after') - } if (editor.isFocused()) { const cursorBlock = editor.getTextCursorPosition().block return editor.insertBlocks(blocks, cursorBlock, 'after') @@ -302,50 +297,42 @@ export default function EditorView() { if (firstBlock) editor.insertBlocks(blocks, firstBlock, 'before') }, [editor]) - // Tracks the paragraph block the *current* dictation session is appending - // to, so consecutive recognized chunks concatenate onto one line instead of - // each becoming its own new block (which, without a stable insertion point, - // ends up stacking in reverse order as the cursor never advances). Cleared - // whenever a dictation session isn't active — see the effect below. - const dictationModeRef = useRef(null) - const dictationSessionBlockIdRef = useRef(null) - // The block the cursor was in when the *current* dictation session started - // — captured before dictation moves focus to the mic button (so the note - // editor is no longer "focused" for the rest of the session). Without this, - // every dictated chunk would fall back to appending at the end of the note - // instead of landing where the user had their cursor. - const dictationAnchorBlockIdRef = useRef(null) - + // Whether the user has put the text cursor somewhere in this note's body. + // ProseMirror keeps its selection when the editor blurs, so once they have, + // dictation can keep typing there even though focus moves to the mic button + // (on click, and deliberately once a session starts, so Enter/Space toggle + // it). Set by the editor wrapper's onFocus; reset whenever a note is loaded + // into the editor — see the hydrate effect. + const editorHasCursorRef = useRef(false) + + // Dictated text goes where the note's cursor is, like typing, and each chunk + // continues from where the last one ended. With no cursor placed in the note + // it starts a line at the bottom instead, and leaves the cursor there so the + // rest of the session carries on along that line. const insertDictatedText = useCallback((text: string) => { - const trimmed = text.trim() - if (!trimmed || !editor) return - - const inSession = dictationModeRef.current === 'dictation' - const targetId = inSession ? dictationSessionBlockIdRef.current : null - - if (targetId) { - const existing = editor.getBlock(targetId) - if (existing) { - const priorText = extractPlainText([existing]) - const merged = priorText ? `${priorText} ${trimmed}` : trimmed - editor.updateBlock(targetId, { content: [{ type: 'text', text: merged, styles: {} }] }) - if (editor.isFocused()) editor.setTextCursorPosition(targetId, 'end') - return - } - // Target block was deleted mid-session (e.g. user backspaced it) — fall - // through and re-anchor to a freshly inserted one. - } + const words = normalizeDictatedText(text) + if (!words || !editor) return - const inserted = insertBlocksAtCursor( - [{ type: 'paragraph', content: [{ type: 'text', text: trimmed, styles: {} }] }], - inSession ? dictationAnchorBlockIdRef.current : null, - ) - const newBlock = inserted[0] - if (newBlock) { - if (inSession) dictationSessionBlockIdRef.current = newBlock.id - if (editor.isFocused()) editor.setTextCursorPosition(newBlock.id, 'end') + const hasCursor = editorHasCursorRef.current || editor.isFocused() + if (hasCursor && editor.transact((tr) => insertDictationAtSelection(tr, words))) return + + // No cursor, or it's on a block that can't hold text (e.g. an image): + // start a new paragraph after that block, or at the bottom of the note — + // filling the note's trailing empty line rather than leaving a gap above. + const doc = editor.document + const anchor = hasCursor ? editor.getTextCursorPosition().block : doc[doc.length - 1] + if (!anchor) return + let targetId = anchor.id + if (!hasCursor && anchor.type === 'paragraph' && anchor.content.length === 0 && anchor.children.length === 0) { + editor.updateBlock(anchor, { content: words }) + } else { + const [inserted] = editor.insertBlocks([{ type: 'paragraph', content: words }], anchor, 'after') + if (!inserted) return + targetId = inserted.id } - }, [editor, insertBlocksAtCursor]) + editor.setTextCursorPosition(targetId, 'end') + editorHasCursorRef.current = true + }, [editor]) // Upload an audio blob to /media and return its URL. The filename extension // must match the blob type so the backend accepts it (.webm/.ogg/.mp3 are allowed). @@ -386,32 +373,6 @@ export default function EditorView() { sttProvider, }) - // Clear the dictation session's target block whenever a session isn't - // active, so the next session starts a fresh paragraph rather than - // continuing to append to a stale one. - useEffect(() => { - dictationModeRef.current = dictation.mode - if (dictation.mode !== 'dictation') { - dictationSessionBlockIdRef.current = null - dictationAnchorBlockIdRef.current = null - } - }, [dictation.mode]) - - // Dictation now focuses the mic button as soon as it starts (so Enter/Space - // can toggle it), which means the editor is no longer "focused" by the time - // insertDictatedText runs. So capture the cursor's block *here*, synchronously, - // right before that focus change happens — this is the last moment the editor - // still reports itself focused for a click that's about to start dictation. - const handleDictationToggle = useCallback(() => { - const willStart = dictation.status === 'idle' || dictation.status === 'error' - if (willStart) { - dictationAnchorBlockIdRef.current = editor?.isFocused() - ? editor.getTextCursorPosition().block.id - : null - } - dictation.toggleDictation() - }, [dictation, editor]) - // Upload a recorded video blob to /media and return its URL + stored filename // (the filename is what the async transcription job references). const uploadVideoBlob = useCallback(async (blob: Blob, mimeType: string, baseName: string): Promise<{ url: string; filename: string }> => { @@ -740,6 +701,9 @@ export default function EditorView() { const blocks = isNewNote ? EMPTY_DOCUMENT : parseNoteContent(note?.content ?? '[]') isHydratingEditor.current = true editor.replaceBlocks(editor.document, blocks as Parameters[1]) + // The selection just landed wherever the replaced content left it, not + // somewhere the user chose — unless they're in the editor right now. + editorHasCursorRef.current = editor.isFocused() currentNoteContent.current = extractPlainText(blocks as unknown[]) syncedEditorKey.current = editorKey hasPendingChanges.current = false @@ -1852,7 +1816,7 @@ export default function EditorView() { anchorRef={exportAnchorRef} onPlayPause={handlePlayPause} dictation={dictation} - onDictationToggle={handleDictationToggle} + onDictationToggle={dictation.toggleDictation} onRecordToggle={dictation.toggleRecording} insertMode={ttsInsertMode} onToggleInsertMode={() => setTtsInsertMode((v) => !v)} @@ -1893,7 +1857,13 @@ export default function EditorView() { -
+
{ if (editor.isFocused()) editorHasCursorRef.current = true }} + > setTtsInsertMode((v) => !v)}