From 2496374904f45beb8a69be43b13524800ed61b9c Mon Sep 17 00:00:00 2001 From: arc53-machine <232052973+arc53-machine@users.noreply.github.com> Date: Mon, 28 Sep 2026 13:21:28 +0100 Subject: [PATCH 1/4] Parse chat math in the markdown parser, not with string rewrites Inline \( \) math rendered as a literal $x$ (the rewrite produced single dollars, which remark-math is told to ignore), a one-line \[ \] shrank to inline math, model-written $x$ never rendered, and the citation pass linked [1] inside formulas. - \( \) and \[ \] are micromark constructs: inline anywhere, display when \[ starts a line and its \] ends one. A line such as "\[a\] and more" stays inline instead of swallowing the rest of the answer, which is why this does not alias micromark-extension-llm-math. - $x$ follows Pandoc's rules (ported from LibreChat, MIT), so prices stay text. - A paragraph holding only $$ or \[ math, and anything KaTeX can only set in display mode (\tag, align, ...), becomes display math. - mhchem is loaded for \ce and \pu. - Citations are linked on the syntax tree, so code, math, link text and escaped brackets are never touched. - While streaming, open emphasis, code and $$ / \[ math are closed (remend plus a code-aware math pass) and a formula KaTeX cannot set yet shows in the muted colour. Each top-level block renders on its own and is memoised, so only the growing block is parsed and typeset again. --- frontend/package-lock.json | 22 +- frontend/package.json | 10 +- .../MarkdownAnswer.codeSpans.test.ts | 194 ------ .../MarkdownAnswer.codeSpans.test.tsx | 209 ++++++ .../conversation/MarkdownAnswer.math.test.tsx | 524 +++++++++++++++ frontend/src/conversation/MarkdownAnswer.tsx | 604 +++++++++--------- .../conversation/markdown/answerMarkdown.ts | 241 +++++++ .../markdown/micromarkExtension.ts | 25 + .../conversation/markdown/remarkCitations.ts | 100 +++ .../markdown/remarkDisplayMath.ts | 134 ++++ .../conversation/markdown/singleDollarMath.ts | 160 +++++ frontend/src/conversation/markdown/texMath.ts | 398 ++++++++++++ 12 files changed, 2106 insertions(+), 515 deletions(-) delete mode 100644 frontend/src/conversation/MarkdownAnswer.codeSpans.test.ts create mode 100644 frontend/src/conversation/MarkdownAnswer.codeSpans.test.tsx create mode 100644 frontend/src/conversation/MarkdownAnswer.math.test.tsx create mode 100644 frontend/src/conversation/markdown/answerMarkdown.ts create mode 100644 frontend/src/conversation/markdown/micromarkExtension.ts create mode 100644 frontend/src/conversation/markdown/remarkCitations.ts create mode 100644 frontend/src/conversation/markdown/remarkDisplayMath.ts create mode 100644 frontend/src/conversation/markdown/singleDollarMath.ts create mode 100644 frontend/src/conversation/markdown/texMath.ts diff --git a/frontend/package-lock.json b/frontend/package-lock.json index aee05213..682a6f9f 100644 --- a/frontend/package-lock.json +++ b/frontend/package-lock.json @@ -21,6 +21,8 @@ "lodash": "^4.18.1", "lucide-react": "^1.45.0", "mermaid": "^12.0.0", + "micromark-util-character": "^2.1.1", + "micromark-util-symbol": "^2.0.1", "radix-ui": "^1.6.7", "react": "^19.3.0", "react-chartjs-2": "^5.3.1", @@ -38,7 +40,11 @@ "rehype-katex": "^7.0.1", "remark-gfm": "^4.0.1", "remark-math": "^6.0.0", - "tailwind-merge": "^3.6.0" + "remark-parse": "^11.0.0", + "remend": "^1.3.1", + "tailwind-merge": "^3.6.0", + "unified": "^11.0.5", + "unist-util-visit": "^5.1.0" }, "devDependencies": { "@eslint/js": "^9.39.5", @@ -46,6 +52,7 @@ "@tailwindcss/postcss": "^4.3.3", "@types/d3-force": "^3.0.10", "@types/lodash": "^4.17.25", + "@types/mdast": "^4.0.4", "@types/react": "^19.3.0", "@types/react-dom": "^19.3.0", "@types/react-syntax-highlighter": "^15.5.13", @@ -64,6 +71,7 @@ "happy-dom": "^20.14.5", "husky": "^9.1.7", "lint-staged": "^17.5.1", + "micromark-util-types": "^2.0.3", "postcss": "^8.5.28", "prettier": "^3.9.6", "prettier-plugin-tailwindcss": "^0.8.1", @@ -9910,9 +9918,9 @@ "license": "MIT" }, "node_modules/micromark-util-types": { - "version": "2.0.2", - "resolved": "https://registry.npmjs.org/micromark-util-types/-/micromark-util-types-2.0.2.tgz", - "integrity": "sha512-Yw0ECSpJoViF1qTU4DC6NwtC4aWGt1EkzaQB8KPPyCRR8z9TWeV0HbEFGTO+ZY1wB22zmxnJqhPyTpOVCpeHTA==", + "version": "2.0.3", + "resolved": "https://registry.npmjs.org/micromark-util-types/-/micromark-util-types-2.0.3.tgz", + "integrity": "sha512-oxB2Ik03hI0gv+VNn9tnh1t1YEe9MDPptViAEgfdf3YQHsn0pzGTgCdlSCJXcwhqm8phaEuM7zeEu3QQzVBrPg==", "funding": [ { "type": "GitHub Sponsors", @@ -11277,6 +11285,12 @@ "url": "https://opencollective.com/unified" } }, + "node_modules/remend": { + "version": "1.3.1", + "resolved": "https://registry.npmjs.org/remend/-/remend-1.3.1.tgz", + "integrity": "sha512-N3DiY5qbRPoa5vkxn1oDLMyXOVTeo6Hp+XOj6SIqJAYUgLS0Q587gILPMom/qm86AQ/ZrcOdwEIzCz8V3J0nxQ==", + "license": "Apache-2.0" + }, "node_modules/reselect": { "version": "5.3.0", "resolved": "https://registry.npmjs.org/reselect/-/reselect-5.3.0.tgz", diff --git a/frontend/package.json b/frontend/package.json index 81753fa8..64a9af9b 100644 --- a/frontend/package.json +++ b/frontend/package.json @@ -38,6 +38,8 @@ "lodash": "^4.18.1", "lucide-react": "^1.45.0", "mermaid": "^12.0.0", + "micromark-util-character": "^2.1.1", + "micromark-util-symbol": "^2.0.1", "radix-ui": "^1.6.7", "react": "^19.3.0", "react-chartjs-2": "^5.3.1", @@ -55,7 +57,11 @@ "rehype-katex": "^7.0.1", "remark-gfm": "^4.0.1", "remark-math": "^6.0.0", - "tailwind-merge": "^3.6.0" + "remark-parse": "^11.0.0", + "remend": "^1.3.1", + "tailwind-merge": "^3.6.0", + "unified": "^11.0.5", + "unist-util-visit": "^5.1.0" }, "devDependencies": { "@eslint/js": "^9.39.5", @@ -63,6 +69,7 @@ "@tailwindcss/postcss": "^4.3.3", "@types/d3-force": "^3.0.10", "@types/lodash": "^4.17.25", + "@types/mdast": "^4.0.4", "@types/react": "^19.3.0", "@types/react-dom": "^19.3.0", "@types/react-syntax-highlighter": "^15.5.13", @@ -81,6 +88,7 @@ "happy-dom": "^20.14.5", "husky": "^9.1.7", "lint-staged": "^17.5.1", + "micromark-util-types": "^2.0.3", "postcss": "^8.5.28", "prettier": "^3.9.6", "prettier-plugin-tailwindcss": "^0.8.1", diff --git a/frontend/src/conversation/MarkdownAnswer.codeSpans.test.ts b/frontend/src/conversation/MarkdownAnswer.codeSpans.test.ts deleted file mode 100644 index befd3305..00000000 --- a/frontend/src/conversation/MarkdownAnswer.codeSpans.test.ts +++ /dev/null @@ -1,194 +0,0 @@ -/** - * Code must survive the markdown string pre-passes. - * - * ``processMarkdownContent`` rewrote every ``[N]`` as ``[N](#cite-N)`` and - * ``preprocessLaTeX`` every ``\[…\]`` as ``$$…$$``, neither with any awareness - * of code, so an index subscript or a bash prompt escape came out corrupted in - * the rendered answer — and in what the copy button copied. - * - * Neither rewrite can move into a remark plugin, where a ``code`` node would be - * distinct from a ``text`` one: remark-math tokenises ``$``/``$$`` while - * parsing, so the LaTeX pass has to happen on the raw string. Hence the shared - * ``applyOutsideCode`` masking helper rather than an AST visitor. - */ -import { describe, expect, it } from 'vitest'; - -import { - applyOutsideCode, - preprocessLaTeX, - processMarkdownContent, -} from './MarkdownAnswer'; - -/** The rendered text of a single-segment answer. */ -const text = (source: string) => processMarkdownContent(source)[0].content; - -describe('citation rewriting spares code', () => { - it('leaves an index subscript alone inside a fence', () => { - const source = '```python\nfor row in cur:\n print(row[0])\n```'; - expect(text(source)).toBe(source); - }); - - it('leaves an index subscript alone inside an inline span', () => { - expect(text('Use `items[2]` here.')).toBe('Use `items[2]` here.'); - }); - - it('leaves a mid-sentence triple-backtick span alone', () => { - expect(text('inline ```code[0]``` end')).toBe('inline ```code[0]``` end'); - }); - - it('leaves a triple-backtick span that starts a line alone', () => { - // A backtick fence's info string cannot contain backticks, so this opens a - // code span, not an unclosed fence that swallows the rest of the answer. - expect(text('```code[0]``` and see [1]')).toBe( - '```code[0]``` and see [1](#cite-1)', - ); - }); - - it('closes an inline span only on a run of its own length', () => { - // The `` run does not close the ` span, so `b[1]` is still code. - const source = 'Use `a`` b[1] more` end'; - expect(text(source)).toBe(source); - }); - - it('recognises a fence closed on a CRLF line', () => { - expect(text('```js\r\nconst a = arr[0];\r\n```\r\nSee [1].')).toBe( - '```js\r\nconst a = arr[0];\r\n```\r\nSee [1](#cite-1).', - ); - }); - - it('does not carry an inline span across a blank line', () => { - // The blank line ends the paragraph, so the backticks never pair and - // `bar[1]` is prose — which is how remark parses it too. - expect(text('a `foo\n\nbar[1]` end')).toBe( - 'a `foo\n\nbar[1](#cite-1)` end', - ); - }); - - it('leaves an inline span that wraps a line alone', () => { - expect(text('use `arr[0]\nnext` here')).toBe('use `arr[0]\nnext` here'); - }); - - it('leaves a tilde fence alone', () => { - const source = '~~~python\nprint(row[0])\n~~~'; - expect(text(source)).toBe(source); - }); - - it('leaves a fence nested in a longer one alone', () => { - // A ``` run does not close a ```` fence, so the inner block is content. - const source = '````markdown\n```js\nconst a = arr[0];\n```\n````'; - expect(text(source)).toBe(source); - }); - - it('leaves a fence indented under a list item alone', () => { - const source = - '1. Step:\n\n ```js\n const a = arr[0];\n\n const b = arr[1];\n ```'; - expect(text(source)).toBe(source); - }); - - it('leaves an unterminated fence alone while the answer is still streaming', () => { - const source = '```python\nprint(row[0]'; - expect(text(source)).toBe(source); - }); - - it('leaves an unterminated inline span alone while streaming', () => { - expect(text('Use `items[2')).toBe('Use `items[2'); - }); - - it('leaves mermaid node labels alone', () => { - // The cite pass runs before the ```mermaid split, so diagram source was - // corrupted too and the render failed. - const source = '```mermaid\nflowchart TD\n A[0] --> B[1]\n```'; - expect(processMarkdownContent(source)[0].content).toBe( - 'flowchart TD\n A[0] --> B[1]', - ); - }); - - it('still links a real citation in prose', () => { - expect(text('See source [1] for details.')).toBe( - 'See source [1](#cite-1) for details.', - ); - }); - - it('links prose citations on either side of a fence without touching it', () => { - expect( - text('Per [1], run:\n```js\nconst a = arr[0];\n```\nthen [2].'), - ).toBe( - 'Per [1](#cite-1), run:\n```js\nconst a = arr[0];\n```\nthen [2](#cite-2).', - ); - }); - - it('still links a citation on an indented list continuation line', () => { - // Four-space indents are list content far more often than they are code, - // so they are prose to the masker. - expect(text('- one\n - nested [1]')).toBe( - '- one\n - nested [1](#cite-1)', - ); - }); - - it('leaves an already-linked reference alone', () => { - expect(text('see [1](#cite-1) ok')).toBe('see [1](#cite-1) ok'); - }); -}); - -describe('LaTeX preprocessing spares code', () => { - it('leaves bash prompt escapes alone', () => { - // `\[`/`\]` are non-printing-sequence markers in PS1; the block rule turned - // them into `$$` delimiters and broke the prompt. - const source = '```bash\nPS1="\\[\\e[0m\\]$ "\n```'; - expect(preprocessLaTeX(source)).toBe(source); - }); - - it('leaves bracket character classes alone in an inline span', () => { - expect(preprocessLaTeX('match `\\[0-9\\]` here')).toBe( - 'match `\\[0-9\\]` here', - ); - }); - - it('still converts real inline math in prose', () => { - expect(preprocessLaTeX('The value \\(x^2\\) is fine.')).toBe( - 'The value $x^2$ is fine.', - ); - }); - - it('still converts real block math in prose', () => { - expect(preprocessLaTeX('Given \\[a+b\\] we get')).toBe( - 'Given $$a+b$$ we get', - ); - }); - - it('does not link a subscript inside the math it just produced', () => { - // `$a[1](#cite-1)$` is not something KaTeX can render. - expect(text('value \\(a[1]\\) end')).toBe('value $a[1]$ end'); - expect(text('given \\[a[1]+b\\] we get [2]')).toBe( - 'given $$a[1]+b$$ we get [2](#cite-2)', - ); - }); -}); - -describe('applyOutsideCode', () => { - const shout = (segment: string) => segment.toUpperCase(); - - it('transforms prose either side of every code span', () => { - expect(applyOutsideCode('a `b` c ```d``` e', shout)).toBe( - 'A `b` C ```d``` E', - ); - }); - - it('handles several inline spans on one line', () => { - expect(applyOutsideCode('x `a` y `b` z', shout)).toBe('X `a` Y `b` Z'); - }); - - it('treats a double-backtick span containing a backtick as one span', () => { - expect(applyOutsideCode('x ``a`b`` y', shout)).toBe('X ``a`b`` Y'); - }); - - it('does not close a long fence on a shorter run', () => { - expect(applyOutsideCode('````\na\n```\nb\n````\nc', shout)).toBe( - '````\na\n```\nb\n````\nC', - ); - }); - - it('passes an empty string through', () => { - expect(applyOutsideCode('', shout)).toBe(''); - }); -}); diff --git a/frontend/src/conversation/MarkdownAnswer.codeSpans.test.tsx b/frontend/src/conversation/MarkdownAnswer.codeSpans.test.tsx new file mode 100644 index 00000000..835a029e --- /dev/null +++ b/frontend/src/conversation/MarkdownAnswer.codeSpans.test.tsx @@ -0,0 +1,209 @@ +/** + * Code must survive citation linking and math parsing. + * + * Both used to be regex passes over the raw answer, and both corrupted the + * user's own code: `print(row[0])` rendered as `print(row[0](#cite-0))`, and + * `PS1="\[\e[0m\]"` as `PS1="$$\e[0m$$"`, in the answer and in what the copy + * button copied. Math is now parsed by the markdown parser and citations are + * linked on its syntax tree, where code is never plain text; these cases pin + * that down on the rendered answer. + */ +import { act } from 'react'; +import { createRoot, type Root } from 'react-dom/client'; + +vi.mock('../hooks', () => ({ useDarkTheme: () => [false, vi.fn()] })); +vi.mock('../components/MermaidRenderer', () => ({ + default: ({ code }: { code: string }) => ( +
{code}
+ ), +})); +vi.mock('../components/CopyButton', () => ({ default: () => null })); + +import MarkdownAnswer from './MarkdownAnswer'; + +Object.assign(globalThis, { IS_REACT_ACT_ENVIRONMENT: true }); + +let container: HTMLDivElement; +let root: Root; + +beforeEach(() => { + container = document.createElement('div'); + document.body.appendChild(container); + root = createRoot(container); +}); + +afterEach(() => { + act(() => root.unmount()); + container.remove(); +}); + +function render(content: string, isStreaming = false) { + act(() => + root.render(), + ); +} + +/** Text of every code block and inline code span, in order. */ +function code() { + return Array.from(container.querySelectorAll('code')).map( + (node) => node.textContent, + ); +} + +/** Citation pills, the only buttons these answers render. */ +function citations() { + return Array.from(container.querySelectorAll('button')).map( + (node) => node.textContent, + ); +} + +function math() { + return container.querySelectorAll('.katex, .katex-error').length; +} + +describe('citation linking spares code', () => { + it('leaves an index subscript alone inside a fence', () => { + render('```python\nfor row in cur:\n print(row[0])\n```'); + expect(code()).toEqual(['for row in cur:\n print(row[0])']); + expect(citations()).toEqual([]); + }); + + it('leaves an index subscript alone inside an inline span', () => { + render('Use `items[2]` here.'); + expect(code()).toEqual(['items[2]']); + expect(citations()).toEqual([]); + }); + + it('leaves a mid-sentence triple-backtick span alone', () => { + render('inline ```code[0]``` end'); + expect(code()).toEqual(['code[0]']); + expect(citations()).toEqual([]); + }); + + it('leaves a triple-backtick span that starts a line alone', () => { + // A backtick fence's info string cannot contain backticks, so this opens a + // code span, not an unclosed fence that swallows the rest of the answer. + render('```code[0]``` and see [1]'); + expect(code()).toEqual(['code[0]']); + expect(citations()).toEqual(['1']); + }); + + it('closes an inline span only on a run of its own length', () => { + // The `` run does not close the ` span, so `b[1]` is still code. + render('Use `a`` b[1] more` end'); + expect(code()).toEqual(['a`` b[1] more']); + expect(citations()).toEqual([]); + }); + + it('recognises a fence closed on a CRLF line', () => { + render('```js\r\nconst a = arr[0];\r\n```\r\nSee [1].'); + expect(code()).toEqual(['const a = arr[0];']); + expect(citations()).toEqual(['1']); + }); + + it('does not carry an inline span across a blank line', () => { + // The blank line ends the paragraph, so the backticks never pair and + // `bar[1]` is prose. + render('a `foo\n\nbar[1]` end'); + expect(code()).toEqual([]); + expect(citations()).toEqual(['1']); + }); + + it('leaves an inline span that wraps a line alone', () => { + render('use `arr[0]\nnext` here'); + expect(code()).toEqual(['arr[0] next']); + expect(citations()).toEqual([]); + }); + + it('leaves a tilde fence alone', () => { + render('~~~python\nprint(row[0])\n~~~'); + expect(code()).toEqual(['print(row[0])']); + expect(citations()).toEqual([]); + }); + + it('leaves a fence nested in a longer one alone', () => { + // A ``` run does not close a ```` fence, so the inner block is content. + render('````markdown\n```js\nconst a = arr[0];\n```\n````'); + expect(code()).toEqual(['```js\nconst a = arr[0];\n```']); + expect(citations()).toEqual([]); + }); + + it('leaves a fence indented under a list item alone', () => { + render( + '1. Step:\n\n ```js\n const a = arr[0];\n\n const b = arr[1];\n ```', + ); + expect(code()).toEqual(['const a = arr[0];\n\nconst b = arr[1];']); + expect(citations()).toEqual([]); + }); + + it('leaves an unterminated fence alone while the answer is still streaming', () => { + render('```python\nprint(row[0]', true); + expect(code()).toEqual(['print(row[0]']); + expect(citations()).toEqual([]); + }); + + it('leaves an unterminated inline span alone while streaming', () => { + render('Use `items[2', true); + expect(code()).toEqual(['items[2']); + expect(citations()).toEqual([]); + }); + + it('leaves mermaid node labels alone', () => { + render('```mermaid\nflowchart TD\n A[0] --> B[1]\n```'); + expect( + container.querySelector('[data-testid="mermaid"]')?.textContent, + ).toBe('flowchart TD\n A[0] --> B[1]'); + expect(citations()).toEqual([]); + }); + + it('still links a real citation in prose', () => { + render('See source [1] for details.'); + expect(citations()).toEqual(['1']); + expect(container.textContent).toBe('See source 1 for details.'); + }); + + it('links prose citations on either side of a fence without touching it', () => { + render('Per [1], run:\n```js\nconst a = arr[0];\n```\nthen [2].'); + expect(code()).toEqual(['const a = arr[0];']); + expect(citations()).toEqual(['1', '2']); + }); + + it('still links a citation on an indented list continuation line', () => { + render('- one\n - nested [1]'); + expect(citations()).toEqual(['1']); + }); + + it('leaves an already-linked reference alone', () => { + render('see [1](#cite-1) ok'); + expect(citations()).toEqual(['1']); + }); +}); + +describe('math parsing spares code', () => { + it('leaves bash prompt escapes alone', () => { + // `\[`/`\]` are non-printing-sequence markers in PS1; the old block rule + // turned them into `$$` delimiters and broke the prompt. + render('```bash\nPS1="\\[\\e[0m\\]$ "\n```'); + expect(code()).toEqual(['PS1="\\[\\e[0m\\]$ "']); + expect(math()).toBe(0); + }); + + it('leaves bracket character classes alone in an inline span', () => { + render('match `\\[0-9\\]` here'); + expect(code()).toEqual(['\\[0-9\\]']); + expect(math()).toBe(0); + }); + + it('leaves dollars alone in an inline span', () => { + render('Run `echo $HOME and $$` then `$x$`.'); + expect(code()).toEqual(['echo $HOME and $$', '$x$']); + expect(math()).toBe(0); + }); + + it('does not link a subscript inside math', () => { + render('value \\(a[1]\\) end, given \\[a[1]+b\\] we get [2]'); + expect(math()).toBe(2); + expect(container.querySelectorAll('.katex-error').length).toBe(0); + expect(citations()).toEqual(['2']); + }); +}); diff --git a/frontend/src/conversation/MarkdownAnswer.math.test.tsx b/frontend/src/conversation/MarkdownAnswer.math.test.tsx new file mode 100644 index 00000000..c52cf656 --- /dev/null +++ b/frontend/src/conversation/MarkdownAnswer.math.test.tsx @@ -0,0 +1,524 @@ +/** + * Math must render as math, and only math. + * + * ``processMarkdownContent`` rewrote inline ``\(x\)`` to ``$x$`` while + * remark-math runs with ``singleDollarTextMath: false``, so every inline span + * rendered as the literal text ``$x$`` (students asked "what does $ mean"). + * A one-line ``\[ … \]`` became ``$$ … $$`` mid-paragraph, which remark-math + * reads as *inline* math, so display equations shrank into the sentence. And a + * citation pass over the raw string turned ``$$a[1]$$`` into a link inside the + * formula. Math is now parsed by the markdown parser itself; these tests only + * look at what renders. + */ +import { act } from 'react'; +import { createRoot, type Root } from 'react-dom/client'; + +// Records the markdown each ReactMarkdown instance renders. +const markdownRenders = vi.hoisted(() => vi.fn()); +vi.mock('react-markdown', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + default: (props: Parameters[0]) => { + markdownRenders(props.children); + return actual.default(props); + }, + }; +}); +vi.mock('../hooks', () => ({ useDarkTheme: () => [false, vi.fn()] })); +vi.mock('../components/MermaidRenderer', () => ({ + default: ({ code }: { code: string }) => ( +
{code}
+ ), +})); +vi.mock('../components/CopyButton', () => ({ default: () => null })); + +import MarkdownAnswer from './MarkdownAnswer'; + +Object.assign(globalThis, { IS_REACT_ACT_ENVIRONMENT: true }); + +let container: HTMLDivElement; +let root: Root; + +beforeEach(() => { + container = document.createElement('div'); + document.body.appendChild(container); + root = createRoot(container); +}); + +afterEach(() => { + act(() => root.unmount()); + container.remove(); +}); + +type RenderOptions = { isStreaming?: boolean; sourceCount?: number }; + +function render(content: string, options: RenderOptions = {}) { + act(() => root.render()); + const all = container.querySelectorAll('.katex').length; + const display = container.querySelectorAll('.katex-display').length; + const clone = container.cloneNode(true) as HTMLElement; + clone + .querySelectorAll('.katex, .katex-error') + .forEach((node) => node.remove()); + return { + inline: all - display, + display, + errors: container.querySelectorAll('.katex-error').length, + text: clone.textContent ?? '', + }; +} + +/** TeX source of every rendered formula, in order. */ +function formulas() { + return Array.from( + container.querySelectorAll( + '.katex annotation[encoding="application/x-tex"]', + ), + ).map((node) => node.textContent); +} + +/** Citation pills, the only buttons these answers render. */ +function citations() { + return Array.from(container.querySelectorAll('button')).map( + (node) => node.textContent, + ); +} + +describe('backslash delimiters', () => { + it('renders inline \\( \\) math as inline KaTeX, not a literal $', () => { + const out = render('The velocity is \\(v = u + at\\) at time t.'); + expect(out.inline).toBe(1); + expect(out.display).toBe(0); + expect(out.text).not.toContain('$'); + expect(out.text).toContain('at time t.'); + }); + + it.each([ + ['at the start of a line', '\\(v\\) is the final velocity.'], + ['in a list item', '- speed \\(v\\) in metres'], + ['in a table cell', '| q | f |\n|---|---|\n| v | \\(v=u+at\\) |'], + ['wrapping a line', 'where \\(a +\nb\\) holds'], + ['in a heading', '## Speed \\(v\\)'], + ['inside bold', '**the speed \\(v\\)**'], + ])('renders inline math %s', (_, source) => { + const out = render(source); + expect(out.inline).toBe(1); + expect(out.errors).toBe(0); + expect(out.text).not.toContain('$'); + }); + + it('keeps a line break inside inline math as whitespace', () => { + // `\alpha` followed by a newline must not fuse into `\alphab`. + render('then \\(\\alpha\nb\\) holds'); + expect(formulas()).toEqual(['\\alpha\nb']); + }); + + it('renders a dollar sign inside inline math without a KaTeX error', () => { + const out = render('cost \\(C = \\$5 n\\) here'); + expect(out.inline).toBe(1); + expect(out.errors).toBe(0); + }); + + it('renders a one-line \\[ \\] on its own line as display math', () => { + const out = render( + 'Displacement:\n\\[ s = ut + \\tfrac12 at^2 \\]\nwhere u is speed.', + ); + expect(out.display).toBe(1); + expect(out.inline).toBe(0); + expect(out.text).toContain('Displacement:'); + expect(out.text).toContain('where u is speed.'); + }); + + it('renders a one-line \\[ \\] inside a list item as display math', () => { + const out = render( + '1. Step one:\n\n \\[ v^2 = u^2 + 2as \\]\n\n2. Step two', + ); + expect(out.display).toBe(1); + expect(container.querySelectorAll('ol > li').length).toBe(2); + }); + + it('renders multi-line \\[ \\] as display math', () => { + expect(render('Then:\n\n\\[\ns = ut\n\\]\n\nDone.').display).toBe(1); + expect(render('\\[\na+b=c\n\\]').display).toBe(1); + }); + + it('does not read math lines as markdown structure', () => { + // A lone `=` under a line is a setext heading and a `- ` line a list item, + // outside math. + const out = render('\\[\nx\n=\ny\n- z\n\\]'); + expect(out.display).toBe(1); + expect(container.querySelector('h1, ul')).toBeNull(); + expect(formulas()).toEqual(['x\n=\ny\n- z']); + }); + + it('renders mid-sentence \\[ \\] as math', () => { + const out = render('So \\[ s = ut \\] holds.'); + expect(out.inline + out.display).toBe(1); + expect(out.text).toContain('holds.'); + }); + + it('keeps the rest of the answer when a line opens with \\[ \\] and goes on', () => { + // A line-start `\[` is a display block only when its `\]` ends the line; + // otherwise it must not swallow everything after it. + const out = render('\\[a+b=c\\] and more text\n\nNext paragraph.'); + expect(out.inline).toBe(1); + expect(out.display).toBe(0); + expect(out.text).toContain('and more text'); + expect(container.querySelectorAll('p').length).toBe(2); + }); + + it.each([ + ['\\[x=1\\tag{1}\\]', '\\tag'], + ['\\[\\begin{aligned}x&=1\\end{aligned}\\]', 'aligned'], + ['where \\(x = 1 \\tag{2}\\) holds', 'inline \\tag'], + ])('renders %s as display math', (source) => { + const out = render(source); + expect(out.display).toBe(1); + expect(out.errors).toBe(0); + }); + + it('renders display math inside a blockquote', () => { + const out = render('> \\[\n> x\n> =\n> y\n> \\]'); + expect(out.display).toBe(1); + expect(container.querySelector('blockquote .katex-display')).not.toBeNull(); + expect(formulas()).toEqual(['x\n=\ny']); + }); + + it('renders display math inside a list item', () => { + const out = render('- \\[\n x\n =\n y\n \\]'); + expect(out.display).toBe(1); + expect(container.querySelector('li .katex-display')).not.toBeNull(); + }); + + it('does not close a display block across the end of a blockquote', () => { + const out = render('> \\[\n> x\n> =\n\ny\n\\]\n\nNext section\n='); + expect(out.inline + out.display + out.errors).toBe(0); + }); + + it('does not render an unclosed \\( as math', () => { + const out = render('Using \\(v = u'); + expect(out.inline + out.display + out.errors).toBe(0); + }); + + it('does not render an unclosed \\[ block as math', () => { + const out = render('Then\n\n\\[\nv = u\n\nand the rest.'); + expect(out.inline + out.display + out.errors).toBe(0); + expect(out.text).toContain('and the rest.'); + }); + + it('keeps an escaped task checkbox as text', () => { + const out = render('- \\[ \\] todo\n- done'); + expect(out.inline + out.display + out.errors).toBe(0); + expect(out.text).toContain('todo'); + expect(container.querySelectorAll('li').length).toBe(2); + }); + + it('keeps escaped index brackets as text', () => { + const out = render('arr\\[0\\] and arr\\[1\\]'); + expect(out.inline + out.display).toBe(0); + expect(out.text).toContain('arr[0] and arr[1]'); + }); + + it('keeps nested \\( \\) in one span', () => { + // KaTeX cannot typeset a nested `\(`, but the span must not end at the + // inner `\)` and leak the outer one into the text. + const out = render('\\(a + \\(b + c\\)\\) end'); + expect(out.inline + out.errors).toBe(1); + expect(out.text.trim()).toBe('end'); + }); + + it('does not mistake a \\[ \\] after inline code for its own line', () => { + const out = render('Use `f` \\[x\\]\nnext'); + expect(out.inline + out.display).toBe(1); + expect(out.text).not.toContain('$'); + }); + + it('leaves \\( \\) inside code untouched', () => { + const out = render('Type `\\(x\\)` literally.'); + expect(out.inline).toBe(0); + expect(container.querySelector('code')?.textContent).toBe('\\(x\\)'); + }); + + it('leaves bash prompt escapes in a fence untouched', () => { + const out = render('```bash\nPS1="\\[\\e[0m\\]$ "\n```'); + expect(out.inline + out.display).toBe(0); + expect(container.textContent).toContain('PS1="\\[\\e[0m\\]$ "'); + }); +}); + +describe('dollar delimiters', () => { + it.each([ + 'from $2bn to at least $4bn per operation starting Sept 9.', + 'Price is $50 and $100', + '$50 is $20 + $30', + 'Total: $29.50 plus tax', + 'Revenue: $5M to $10M, funding: $1.5B, price: $5K', + '$250k is 25% of $1M', + 'Bitcoin: $0.00001234, Gas: $3.999, Rate: $1.234567890', + 'The total is $1157.90 (existing) + $500 (new investment) = $1657.90.', + 'a $100-$200 range', + 'a $100–$200 range', + 'in the $10k-$20k band', + '- **Total Savings**: $500 + $200 + $150 = $850', + 'Cela coûte 100$ et 200$ en Europe', + 'A single $ sign should not be converted', + 'The price hit $79,455 on', + 'Currency $100 and\nthen $200 later', + 'The cost is $', + 'Use $variable in the code', + '| Date | Price | Note |\n|---|---|---|\n| Aug 21 | $77,300, peak $79,455 | White House Clarity Act push |', + 'It costs $5 and $10 in total.', + 'Plans range between $5-$10 per seat.', + 'Price: $19.99/month or $199/year.', + 'Revenue grew from $3.2B to $4.1B.', + 'Price is $5 (was $10).', + '| Plan | Price |\n|---|---|\n| Basic | $5 |\n| Pro | $10 |', + ])('leaves currency as text: %s', (source) => { + const out = render(source); + expect(out.inline + out.display + out.errors).toBe(0); + const amount = source.match(/\$\d[\d.,]*|\d+\$/); + if (amount) expect(out.text).toContain(amount[0]); + }); + + it.each([ + ['Inline math: $x^2 + y^2 = z^2$', 1], + ['the answer is $3$.', 1], + ['an eigenvalue of $-1$ is expected', 1], + ['the $n$th term', 1], + ['The set is defined as $\\{x | x > 0\\}$.', 1], + ['Calculate $\\text{Total} = \\$500 + \\$200$', 1], + ['The equation $12 \\times 12 = 144$ is simple', 1], + ['Price $100 then equation $x + y = z$ then another price $50', 1], + ['Formula $x^2$ costs $25', 1], + ])('renders %s as math', (source, count) => { + const out = render(source); + expect(out.inline).toBe(count); + expect(out.errors).toBe(0); + }); + + it('keeps the price beside a formula', () => { + render('Cost is $5 per unit and $x$ units.'); + expect(formulas()).toEqual(['x']); + expect(container.textContent).toContain('$5 per unit'); + }); + + it('does not form emphasis out of math', () => { + const out = render('terms $a_1 + b_2$ and $c_{i}^{*}$ here'); + expect(out.inline).toBe(2); + expect(container.querySelector('em, strong')).toBeNull(); + }); + + it('renders chemistry with mhchem', () => { + const out = render('$\\ce{H2O}$ and $\\pu{123 J}$'); + expect(out.inline).toBe(2); + expect(out.errors).toBe(0); + }); + + it.each([ + [ + 'a code span', + 'The error "invalid $lookup namespace" occurs when using `$lookup` operator', + ], + ['escaped dollars', 'Already escaped \\$50 and \\$100'], + ['a line break', 'This has $x\ny$ which spans lines'], + ['an unbalanced brace', 'weird $a}b$ y'], + [ + 'shell PIDs in a fence', + '```bash\npstree -p $$\necho $$\n```\n\nSome text after.', + ], + ])('renders no math across %s', (_, source) => { + const out = render(source); + expect(out.inline + out.display + out.errors).toBe(0); + }); + + it('renders math only outside a fence', () => { + const out = render('```\n$100\n$variable\n```\n\nOutside $x^2$'); + expect(out.inline).toBe(1); + expect(container.querySelector('code')?.textContent).toContain('$variable'); + }); + + it('renders a $$ paragraph as display math', () => { + const out = render('text\n\n$$E=mc^2$$\n\nmore'); + expect(out.display).toBe(1); + expect(out.text).toContain('more'); + }); + + it('keeps mid-line $$ math inline', () => { + const out = render('so $$E=mc^2$$ holds'); + expect(out.inline).toBe(1); + expect(out.display).toBe(0); + }); + + it('renders $$ with \\begin on the fence line', () => { + const out = render( + '$$\\begin{aligned}\na&=b\\\\\nc&=d\n\\end{aligned}$$\n\nAfter.', + ); + expect(out.display).toBe(1); + expect(out.errors).toBe(0); + expect(out.text).toContain('After.'); + }); +}); + +describe('citations', () => { + it('links citations next to inline math and renders the math', () => { + const out = render('Newton: \\(F=ma\\) [1] and \\(p=mv\\)[2].'); + expect(out.inline).toBe(2); + expect(citations()).toEqual(['1', '2']); + }); + + it('never links a bracket inside math', () => { + const out = render('$$a[1]$$ and [3]'); + expect(formulas()).toEqual(['a[1]']); + expect(out.errors).toBe(0); + expect(citations()).toEqual(['3']); + }); + + it('never links a bracket inside single-dollar math', () => { + const out = render('value $a[1]$ end [2]'); + expect(citations()).toEqual(['2']); + expect(out.errors).toBe(0); + }); + + it.each([ + ['**[1]** and [2][3]', ['1', '2', '3']], + ['see `row[0]` and [4]', ['4']], + ['[link](https://x.com) [1]', ['1']], + ['- one\n - nested [1]', ['1']], + ['Per [1], run:\n```js\nconst a = arr[0];\n```\nthen [2].', ['1', '2']], + ])('links %s', (source, expected) => { + render(source); + expect(citations()).toEqual(expected); + }); + + it('leaves code and diagram labels alone', () => { + render('```python\nprint(row[0])\n```'); + expect(container.textContent).toContain('print(row[0])'); + expect(citations()).toEqual([]); + }); + + it('keeps a reference-style link a link', () => { + render('See [the docs][1].\n\n[1]: https://example.com'); + expect( + container.querySelector('a[href="https://example.com"]'), + ).not.toBeNull(); + expect(citations()).toEqual([]); + }); + + it('links only citations within the number of sources', () => { + render('Per [1], [2] and [0] and [7].', { sourceCount: 2 }); + expect(citations()).toEqual(['1', '2']); + expect(container.textContent).toContain('[0] and [7]'); + }); + + it('links no citation when the answer has no sources', () => { + render('Per [1].', { sourceCount: 0 }); + expect(citations()).toEqual([]); + expect(container.textContent).toContain('[1]'); + }); +}); + +describe('mermaid', () => { + it('renders a closed mermaid fence as a diagram, source untouched', () => { + render( + 'Intro [1]\n\n```mermaid\nflowchart TD\n A[0] --> B[1]\n```\n\nOutro', + ); + expect( + container.querySelector('[data-testid="mermaid"]')?.textContent, + ).toBe('flowchart TD\n A[0] --> B[1]'); + expect(citations()).toEqual(['1']); + expect(container.textContent).toContain('Outro'); + }); + + it('shows an unclosed mermaid fence as code while it streams', () => { + render('```mermaid\nflowchart TD\n A --> B', { isStreaming: true }); + expect(container.querySelector('[data-testid="mermaid"]')).toBeNull(); + }); +}); + +describe('streaming', () => { + it('closes an open $$ block while streaming', () => { + const out = render('$$\nx = 1\ny = 2', { isStreaming: true }); + expect(out.display).toBe(1); + }); + + it('closes an open inline $$ while streaming', () => { + const out = render('Text with $$formula', { isStreaming: true }); + expect(out.inline).toBe(1); + expect(out.text).not.toContain('$'); + }); + + it('closes an open \\[ block while streaming', () => { + const out = render('Then:\n\n\\[\nv = u', { isStreaming: true }); + expect(out.display).toBe(1); + }); + + it('closes open bold before an open formula, not after it', () => { + // Appended to the end, `**` turned the closing fence into `$$**`. + const out = render('Energy is **the work it takes\n$$\nE = mc', { + isStreaming: true, + }); + expect(out.display).toBe(1); + expect(out.errors).toBe(0); + expect(formulas()).toEqual(['E = mc']); + }); + + it('holds back an inline $$ with nothing after it yet', () => { + const out = render('Then **bold** and $$', { isStreaming: true }); + expect(out.inline + out.display + out.errors).toBe(0); + expect(out.text).not.toContain('$'); + }); + + it('does not render an unclosed \\( while streaming', () => { + const out = render('Using \\(v = u', { isStreaming: true }); + expect(out.inline + out.display + out.errors).toBe(0); + }); + + it('leaves a trailing $ alone', () => { + const out = render('The cost is $', { isStreaming: true }); + expect(out.inline + out.display).toBe(0); + expect(out.text).toContain('The cost is $'); + }); + + it('closes open bold while streaming', () => { + render('Text **bold', { isStreaming: true }); + expect(container.querySelector('strong')?.textContent).toBe('bold'); + }); + + it('does not add $$ inside an open code fence', () => { + render('```bash\necho $$', { isStreaming: true }); + expect(container.querySelector('code')?.textContent).toBe('echo $$'); + }); + + it('does not heal once the answer is complete', () => { + render('Text **bold'); + expect(container.querySelector('strong')).toBeNull(); + }); + + it('shows a formula that cannot render yet in the muted colour', () => { + const out = render('$$\n\\frac{a}{', { isStreaming: true }); + expect(out.errors).toBe(1); + const error = container.querySelector('.katex-error'); + expect(error?.getAttribute('style')).toContain('var(--muted-foreground)'); + }); +}); + +describe('block rendering', () => { + it('re-renders only the block that changed', () => { + render('First \\(x^2\\) [1].\n\nSecond', { isStreaming: true }); + markdownRenders.mockClear(); + render('First \\(x^2\\) [1].\n\nSecond paragraph grows', { + isStreaming: true, + }); + // The first paragraph, formula and all, is not parsed or typeset again. + expect(markdownRenders.mock.calls).toEqual([['Second paragraph grows']]); + expect(container.textContent).toContain('Second paragraph grows'); + }); + + it('keeps footnotes working across blocks', () => { + render('Claim[^1].\n\nMore.\n\n[^1]: The note.'); + expect( + container.querySelector('a[href="#user-content-fn-1"]'), + ).not.toBeNull(); + }); +}); diff --git a/frontend/src/conversation/MarkdownAnswer.tsx b/frontend/src/conversation/MarkdownAnswer.tsx index 8736645b..39d4bedb 100644 --- a/frontend/src/conversation/MarkdownAnswer.tsx +++ b/frontend/src/conversation/MarkdownAnswer.tsx @@ -1,16 +1,17 @@ import 'katex/dist/katex.min.css'; +// Registers `\ce` and `\pu` with the KaTeX that rehype-katex renders with. +import 'katex/contrib/mhchem'; -import { Fragment, type ReactNode, useMemo } from 'react'; +import { Fragment, memo, type ReactNode, useMemo } from 'react'; import { useTranslation } from 'react-i18next'; -import ReactMarkdown from 'react-markdown'; +import ReactMarkdown, { type Components } from 'react-markdown'; import { Prism as SyntaxHighlighter } from 'react-syntax-highlighter'; import { oneLight, vscDarkPlus, } from 'react-syntax-highlighter/dist/cjs/styles/prism'; import rehypeKatex from 'rehype-katex'; -import remarkGfm from 'remark-gfm'; -import remarkMath from 'remark-math'; +import type { PluggableList } from 'unified'; import { markdownHeadings } from '@/lib/markdown'; @@ -19,6 +20,14 @@ import MermaidRenderer from '../components/MermaidRenderer'; import { Button } from '../components/ui/button'; import { useDarkTheme } from '../hooks'; import classes from './ConversationBubble.module.css'; +import { + answerSyntaxPlugins, + healStreamingMarkdown, + normalizeMathFences, + splitAnswerBlocks, +} from './markdown/answerMarkdown'; +import { remarkCitations } from './markdown/remarkCitations'; +import { remarkDisplayMath } from './markdown/remarkDisplayMath'; import { resolveSandboxLink, type SandboxArtifact, @@ -26,118 +35,62 @@ import { } from './sandboxLinks'; import { cn } from '@/lib/utils'; -// One fenced block or inline code span. Backtick runs are length-matched, so a -// ```` fence closes only on ```` and nested fences stay masked. The -// unterminated alternatives keep an open span matched too: answers stream in, -// so a fence is open for most of its life and needs protecting the whole time, -// not just once the closing fence arrives. A backtick fence's info string may -// not itself contain backticks, so a line that opens with an inline span stays -// an inline span. Four-space indented blocks are deliberately not masked — -// list continuation lines are indented the same way, and masking those would -// drop real citations out of nested lists. -const CODE_SPAN = - /(? string, - onMasked: (segment: string) => string = (segment) => segment, -): string { - const pattern = new RegExp(mask.source, mask.flags); // its own lastIndex - let result = ''; - let index = 0; - let match: RegExpExecArray | null; - - while ((match = pattern.exec(content)) !== null) { - result += transform(content.slice(index, match.index)) + onMasked(match[0]); - index = pattern.lastIndex; - } - return result + transform(content.slice(index)); -} - -// Runs `transform` over the prose in `content`, leaving code untouched. -// -// The rewrites below are plain regex over the whole document, and both used to -// corrupt the user's own code: `print(row[0])` rendered as -// `print(row[0](#cite-0))`, and `PS1="\[\e[0m\]"` as `PS1="$$\e[0m$$"`. Neither -// can move into a remark plugin: remark-math tokenises `$`/`$$` while parsing, -// so the LaTeX pass has to run on the raw string before remark ever sees it. -export function applyOutsideCode( - content: string, - transform: (segment: string) => string, -): string { - return transformOutside(content, CODE_SPAN, transform); -} - -// Rewrites one LaTeX span into the ``$$``/``$`` form remark-math parses. -function toDollarMath(span: string): string { - const equation = span.slice(2, -2); - return span.startsWith('\\[') ? `$$${equation}$$` : `$${equation}$`; -} - -// Replaces block-level ``\[ \]`` and inline ``\( \)`` LaTeX delimiters with the -// ``$$``/``$`` forms remark-math understands. -export function preprocessLaTeX(content: string): string { - return applyOutsideCode(content, (prose) => - prose.replace(LATEX_SPAN, toDollarMath), +// One top-level block of the answer. Memoised, so while an answer streams only +// the block that is still growing is parsed and typeset again. +const MarkdownBlock = memo(function MarkdownBlock({ + content, + remarkPlugins, + components, +}: { + content: string; + remarkPlugins: PluggableList; + components: Components; +}) { + return ( + + {content} + ); -} +}); -// Turns citation references ``[N]`` into ``[N](#cite-N)`` links so -// ReactMarkdown renders them as tags we can style. The lookarounds skip -// references that are already links. -function linkCitations(prose: string): string { - return prose.replace( - /(? `[${num}](#cite-${num})`, +// Artifact lists are rebuilt on every render of the bubble; keep one identity +// per content so the memoised blocks are not re-rendered for nothing. +function useStableArtifacts(list?: SandboxArtifact[]) { + const signature = JSON.stringify( + list?.map(({ id, label, toolName, ref }) => [id, label, toolName, ref]), ); + return useMemo(() => list, [signature]); } -type ContentSegment = { type: 'text' | 'mermaid'; content: string }; - -export function processMarkdownContent(content: string): ContentSegment[] { - // Citations are linked outside code — an index subscript like `row[0]` is not - // a citation — and outside the math this same pass produces. Math the answer - // already wrote as ``$…$`` stays unmasked, since ``$`` is also a currency - // sign. This has to run before the ```mermaid split below, so diagram source - // needs the same guard. - const processedContent = applyOutsideCode(content, (prose) => - transformOutside(prose, LATEX_SPAN, linkCitations, toDollarMath), - ); - - const contentSegments: ContentSegment[] = []; - let lastIndex = 0; - const regex = /```mermaid\n([\s\S]*?)```/g; - let match; - - while ((match = regex.exec(processedContent)) !== null) { - const textBefore = processedContent.substring(lastIndex, match.index); - if (textBefore) contentSegments.push({ type: 'text', content: textBefore }); - contentSegments.push({ type: 'mermaid', content: match[1].trim() }); - lastIndex = match.index + match[0].length; - } - - const textAfter = processedContent.substring(lastIndex); - if (textAfter) contentSegments.push({ type: 'text', content: textAfter }); - - return contentSegments; -} +type AnswerGroup = + { type: 'markdown'; blocks: string[] } | { type: 'mermaid'; content: string }; export default function MarkdownAnswer({ content, isStreaming, - artifacts, - turnArtifacts, + sourceCount, + artifacts: artifactsProp, + turnArtifacts: turnArtifactsProp, onOpenArtifact, }: { content: string; isStreaming?: boolean; + /** + * How many sources the answer cites from; ``[N]`` beyond it is left as + * text, with no source to jump to. Unset links every ``[N]``. + */ + sourceCount?: number; /** * Every artifact in the conversation, in creation order (``A1`` is the * first) — refs are conversation-scoped, so a link may name an earlier turn's @@ -150,230 +103,249 @@ export default function MarkdownAnswer({ }) { const { t } = useTranslation(); const [isDarkTheme] = useDarkTheme(); - // Re-runs on every streamed token otherwise. - const contentSegments = useMemo( - () => processMarkdownContent(content), - [content], + const artifacts = useStableArtifacts(artifactsProp); + const turnArtifacts = useStableArtifacts(turnArtifactsProp); + + const groups = useMemo(() => { + const normalized = normalizeMathFences(content); + const blocks = splitAnswerBlocks( + isStreaming ? healStreamingMarkdown(normalized) : normalized, + ); + // Consecutive markdown blocks share one column; a diagram breaks it. + const grouped: AnswerGroup[] = []; + for (const block of blocks) { + const last = grouped[grouped.length - 1]; + if (block.type === 'mermaid') grouped.push(block); + else if (last?.type === 'markdown') last.blocks.push(block.content); + else grouped.push({ type: 'markdown', blocks: [block.content] }); + } + return grouped; + }, [content, isStreaming]); + + const remarkPlugins = useMemo( + () => [ + ...answerSyntaxPlugins, + remarkDisplayMath, + [remarkCitations, { sourceCount }], + ], + [sourceCount], ); - // Shared by the `a` and `img` renderers: a generated file is already on the - // turn as a download chip, so both point at the chip rather than at a URL no - // browser can open. - const renderArtifactChip = ( - artifact: SandboxArtifact, - content: ReactNode, - ) => { - if (!onOpenArtifact) return <>{content}; - return ( - + ); + }; + + return { + ...markdownHeadings, + a({ href, children }) { + // A generated file is already on the turn as a download + // chip, but the model links it with a `sandbox:`/`artifact:` + // URL no browser can open. Point the link at the chip + // instead, and never leave a dead anchor behind. + const sandboxLink = resolveSandboxLink(href, artifacts, turnArtifacts); + if (sandboxLink.kind === 'plain') { + return <>{children}; } - /* Sits mid-sentence, so it must wrap with the surrounding text. */ - className="whitespace-normal" - > - {content} - - ); - }; + if (sandboxLink.kind === 'artifact') { + return renderArtifactChip(sandboxLink.artifact, children); + } + if (href?.startsWith('#cite-')) { + const num = href.replace('#cite-', ''); + const sourceIdx = parseInt(num, 10) - 1; + return ( + + ); + } + return ( + + {children} + + ); + }, + img({ src, alt }) { + // `![chart](sandbox:/mnt/data/chart.png)` is how a model + // announces a plot it produced. react-markdown runs + // urlTransform over `src` too, so without this the invented + // scheme survives and renders a broken-image box beside the + // chip that opens the very same file. + const source = typeof src === 'string' ? src : undefined; + const sandboxLink = resolveSandboxLink( + source, + artifacts, + turnArtifacts, + ); + if (sandboxLink.kind === 'artifact') { + const { artifact } = sandboxLink; + return renderArtifactChip( + artifact, + alt || artifact.label || t('conversation.openFile'), + ); + } + if (sandboxLink.kind === 'plain') { + return <>{alt ?? ''}; + } + return {alt}; + }, + code(props) { + const { children, className, node, ref, ...rest } = props; + const match = /language-(\w+)/.exec(className || ''); + const language = match ? match[1] : ''; + + return match ? ( +
+
+ + {language} + + +
+ + {String(children).replace(/\n$/, '')} + +
+ ) : ( + + {children} + + ); + }, + ul({ children }) { + return ( +
    + {children} +
+ ); + }, + ol({ children }) { + return ( +
    + {children} +
+ ); + }, + table({ children }) { + return ( +
+ + {children} +
+
+ ); + }, + thead({ children }) { + return ( + + {children} + + ); + }, + tr({ children }) { + return ( + + {children} + + ); + }, + th({ children }) { + return {children}; + }, + td({ children }) { + return {children}; + }, + }; + }, [t, isDarkTheme, artifacts, turnArtifacts, onOpenArtifact]); return ( <> - {contentSegments.map((segment, index) => ( + {groups.map((group, index) => ( - {segment.type === 'text' ? ( -
- {children}; - } - if (sandboxLink.kind === 'artifact') { - return renderArtifactChip(sandboxLink.artifact, children); - } - if (href?.startsWith('#cite-')) { - const num = href.replace('#cite-', ''); - const sourceIdx = parseInt(num, 10) - 1; - return ( - - ); - } - return ( - - {children} - - ); - }, - img({ src, alt }) { - // `![chart](sandbox:/mnt/data/chart.png)` is how a model - // announces a plot it produced. react-markdown runs - // urlTransform over `src` too, so without this the invented - // scheme survives and renders a broken-image box beside the - // chip that opens the very same file. - const source = typeof src === 'string' ? src : undefined; - const sandboxLink = resolveSandboxLink( - source, - artifacts, - turnArtifacts, - ); - if (sandboxLink.kind === 'artifact') { - const { artifact } = sandboxLink; - return renderArtifactChip( - artifact, - alt || artifact.label || t('conversation.openFile'), - ); - } - if (sandboxLink.kind === 'plain') { - return <>{alt ?? ''}; - } - return ( - {alt} - ); - }, - code(props) { - const { children, className, node, ref, ...rest } = props; - const match = /language-(\w+)/.exec(className || ''); - const language = match ? match[1] : ''; - - return match ? ( -
-
- - {language} - - -
- - {String(children).replace(/\n$/, '')} - -
- ) : ( - - {children} - - ); - }, - ul({ children }) { - return ( -
    - {children} -
- ); - }, - ol({ children }) { - return ( -
    - {children} -
- ); - }, - table({ children }) { - return ( -
- - {children} -
-
- ); - }, - thead({ children }) { - return ( - - {children} - - ); - }, - tr({ children }) { - return ( - - {children} - - ); - }, - th({ children }) { - return {children}; - }, - td({ children }) { - return {children}; - }, - }} - > - {segment.content} -
+ {group.type === 'markdown' ? ( +
+ {group.blocks.map((block, blockIndex) => ( + + ))}
) : (
- +
)} diff --git a/frontend/src/conversation/markdown/answerMarkdown.ts b/frontend/src/conversation/markdown/answerMarkdown.ts new file mode 100644 index 00000000..6994aec2 --- /dev/null +++ b/frontend/src/conversation/markdown/answerMarkdown.ts @@ -0,0 +1,241 @@ +/** + * The string-level steps of rendering an answer, around the markdown parser. + * + * Math, code and citations are told apart by the parser (see `texMath.ts`, + * `singleDollarMath.ts` and the remark plugins beside them). What is left + * here works on the raw string only where the parser cannot help: + * + * - `normalizeMathFences` moves formula text off a ``$$`` fence line, which + * remark-math would read as the fence's metadata; + * - `healStreamingMarkdown` closes what a half-streamed answer has left open; + * - `splitAnswerBlocks` cuts the answer into its top-level blocks, so a + * streaming answer re-renders only the block that is still growing. + */ +import type { Root, RootContent } from 'mdast'; +import remarkGfm from 'remark-gfm'; +import remarkMath from 'remark-math'; +import remarkParse from 'remark-parse'; +import remend from 'remend'; +import { type PluggableList, unified } from 'unified'; + +import { remarkSingleDollarMath } from './singleDollarMath'; +import { remarkTexMath } from './texMath'; + +/** Syntax plugins; every parse of an answer, split or render, uses them. */ +export const answerSyntaxPlugins: PluggableList = [ + remarkGfm, + [remarkMath, { singleDollarTextMath: false }], + remarkTexMath, + remarkSingleDollarMath, +]; + +// A line's container prefix (blockquote markers, indentation) and the rest. +const LINE = /^((?:[ \t]*>)*[ \t]*)([\s\S]*)$/; +const FENCE_OPEN = /^(`{3,}|~{3,})/; +const DOLLAR_FENCE = /^\$\$\s*$/; + +function splitLine(line: string): [prefix: string, rest: string] { + const [, prefix, rest] = LINE.exec(line) ?? ['', '', line]; + return [prefix, rest]; +} + +function closesFence(rest: string, fence: string): boolean { + const run = rest.trim(); + return ( + run.length >= fence.length && run[0] === fence[0] && /^(?:`+|~+)$/.test(run) + ); +} + +/** + * Puts a ``$$`` fence's formula on its own line: ``$$\begin{aligned}`` opens a + * block whose first line remark-math reads as metadata and drops, and a + * closing ``\end{aligned}$$`` never closes it, so the block ran to the end of + * the answer. Code fences are left alone. + */ +export function normalizeMathFences(content: string): string { + if (!content.includes('$$')) return content; + const out: string[] = []; + let fence: string | null = null; + let inMath = false; + + for (const line of content.split('\n')) { + const [prefix, rest] = splitLine(line); + if (fence) { + if (closesFence(rest, fence)) fence = null; + out.push(line); + continue; + } + if (inMath) { + const closing = /^(.*\S)[ \t]*\$\$\s*$/.exec(rest); + if (DOLLAR_FENCE.test(rest)) { + inMath = false; + } else if (closing) { + out.push(prefix + closing[1], `${prefix}$$`); + inMath = false; + continue; + } + out.push(line); + continue; + } + const codeFence = FENCE_OPEN.exec(rest); + const opening = /^\$\$(?!\$)(.*\S)\s*$/.exec(rest); + if (codeFence) { + fence = codeFence[1]; + } else if (DOLLAR_FENCE.test(rest)) { + inMath = true; + } else if (opening && !opening[1].includes('$$')) { + out.push(`${prefix}$$`, prefix + opening[1].trimStart()); + inMath = true; + continue; + } + out.push(line); + } + return out.join('\n'); +} + +// Inline code in a stretch of prose, open spans included, so a `$$` in it is +// not counted as math. +const INLINE_CODE = /(`+)[\s\S]*?(?:\1|$)/g; + +// Splits a half-streamed answer at the math it is in the middle of writing: +// `head` is everything before that formula, `math` the formula closed. A ``$$`` +// or line-start ``\[`` block gets its closing fence, an inline ``$$`` its +// closing ``$$``, and one with nothing after it yet is held back. Inside a code +// fence nothing is touched: a shell's `$$` is not math. +function splitOpenMath(content: string): { head: string; math: string } { + let fence: string | null = null; + let block: { close: string; prefix: string; start: number } | null = null; + let offset = 0; + + for (const line of content.split('\n')) { + const [prefix, rest] = splitLine(line); + if (fence) { + if (closesFence(rest, fence)) fence = null; + } else if (block) { + const closed = + block.close === '$$' ? DOLLAR_FENCE.test(rest) : /\\\]\s*$/.test(rest); + if (closed) block = null; + } else if (FENCE_OPEN.test(rest)) { + fence = FENCE_OPEN.exec(rest)![1]; + } else if (DOLLAR_FENCE.test(rest)) { + block = { close: '$$', prefix, start: offset }; + } else if (rest.startsWith('\\[') && !rest.includes('\\]')) { + block = { close: '\\]', prefix, start: offset }; + } + offset += line.length + 1; + } + if (fence) return { head: content, math: '' }; + if (block) { + return { + head: content.slice(0, block.start), + math: `${content.slice(block.start)}\n${block.prefix}${block.close}`, + }; + } + + const paragraphStart = content.lastIndexOf('\n\n') + 1; + const prose = content + .slice(paragraphStart) + .replace(INLINE_CODE, '') + .replace(/\\\$/g, ''); + const dollars = prose.match(/\$\$/g)?.length ?? 0; + if (dollars % 2 === 0) return { head: content, math: '' }; + const opener = content.lastIndexOf('$$'); + const formula = content.slice(opener); + return { + head: content.slice(0, opener), + math: formula.slice(2).trim() ? `${formula}$$` : '', + }; +} + +const REMEND_OPTIONS = { + // Math is closed above, code-aware; remend's own pass appends `$$` inside + // an open code fence. + katex: false, + inlineKatex: false, + // A half-typed `[1` or `\[` is not a link, and remend would strip its `[`. + links: false, + images: false, + // These two rewrite text that is already complete, so the answer would + // change once it stops streaming. + comparisonOperators: false, + singleTilde: false, +}; + +/** + * Closes what a half-streamed answer leaves open (``**bold``, `` `code``, + * ``$$`` and ``\[`` math), so it renders as it will once complete instead of + * flashing raw markdown. Only ever applied while the answer streams. + */ +export function healStreamingMarkdown(content: string): string { + // Emphasis is closed before the open formula, never after it: appended to + // the closing fence (``$$**``), it would stop the fence from closing. + const { head, math } = splitOpenMath(content); + if (!math) return remend(head, REMEND_OPTIONS); + const lineBreak = head.match(/\n*$/)![0]; + return ( + remend(head.slice(0, head.length - lineBreak.length), REMEND_OPTIONS) + + lineBreak + + math + ); +} + +export type AnswerBlock = + { type: 'markdown'; content: string } | { type: 'mermaid'; content: string }; + +const parser = unified().use(remarkParse).use(answerSyntaxPlugins); + +// Where a block's first line starts, so an indented block keeps its indent. +function lineStart(content: string, offset: number): number { + const start = content.lastIndexOf('\n', offset - 1) + 1; + return /^[ \t]*$/.test(content.slice(start, offset)) ? start : offset; +} + +// A mermaid fence is drawn once it is closed; until then it streams as code. +function isDiagram(node: RootContent, source: string): boolean { + if (node.type !== 'code' || node.lang !== 'mermaid') return false; + const lines = source.trimEnd().split('\n'); + const opening = FENCE_OPEN.exec(lines[0].trim()); + return ( + lines.length > 1 && + opening !== null && + closesFence(lines[lines.length - 1], opening[1]) + ); +} + +/** + * Cuts an answer into its top-level blocks, each rendered on its own, and its + * closed mermaid fences, drawn as diagrams. An answer with link or footnote + * definitions stays whole between diagrams, since a reference resolves only + * against definitions in the same parse. + */ +export function splitAnswerBlocks(content: string): AnswerBlock[] { + const tree: Root = parser.parse(content); + const keepWhole = tree.children.some( + (node) => node.type === 'definition' || node.type === 'footnoteDefinition', + ); + const blocks: AnswerBlock[] = []; + let groupStart: number | null = null; + + for (const node of tree.children) { + const start = lineStart(content, node.position?.start.offset ?? 0); + const end = node.position?.end.offset ?? content.length; + if (node.type === 'code' && isDiagram(node, content.slice(start, end))) { + if (groupStart !== null) { + blocks.push({ + type: 'markdown', + content: content.slice(groupStart, start), + }); + groupStart = null; + } + blocks.push({ type: 'mermaid', content: node.value.trim() }); + } else if (keepWhole) { + groupStart ??= start; + } else { + blocks.push({ type: 'markdown', content: content.slice(start, end) }); + } + } + if (groupStart !== null) { + blocks.push({ type: 'markdown', content: content.slice(groupStart) }); + } + return blocks; +} diff --git a/frontend/src/conversation/markdown/micromarkExtension.ts b/frontend/src/conversation/markdown/micromarkExtension.ts new file mode 100644 index 00000000..471039c2 --- /dev/null +++ b/frontend/src/conversation/markdown/micromarkExtension.ts @@ -0,0 +1,25 @@ +import type { Extension } from 'micromark-util-types'; +import type { Processor } from 'unified'; + +// The tokens `remark-math`'s `mathFromMarkdown` turns into math nodes, which +// the constructs here emit too. +declare module 'micromark-util-types' { + interface TokenTypeMap { + mathFlow: 'mathFlow'; + mathFlowFence: 'mathFlowFence'; + mathFlowFenceSequence: 'mathFlowFenceSequence'; + mathFlowValue: 'mathFlowValue'; + mathText: 'mathText'; + mathTextData: 'mathTextData'; + mathTextSequence: 'mathTextSequence'; + } +} + +/** Registers a micromark syntax extension from inside a remark plugin. */ +export function addMicromarkExtension( + processor: Processor, + extension: Extension, +): void { + const data = processor.data() as { micromarkExtensions?: Extension[] }; + (data.micromarkExtensions ??= []).push(extension); +} diff --git a/frontend/src/conversation/markdown/remarkCitations.ts b/frontend/src/conversation/markdown/remarkCitations.ts new file mode 100644 index 00000000..2c5e70e7 --- /dev/null +++ b/frontend/src/conversation/markdown/remarkCitations.ts @@ -0,0 +1,100 @@ +/** + * Links source citations, ``[N]``, in the prose of an answer. + * + * Runs on the syntax tree, so only plain text is ever linked: code, inline + * code and math are their own nodes, never `text`, and link text is skipped. + * An index like `row[0]` in code, or `a[1]` in a formula, stays what it is. + * A bare ``[1]`` with no matching definition is text to the parser; a + * reference link ``[docs][1]`` with a ``[1]: url`` definition stays a link, + * and brackets the answer escaped, ``arr\[1\]``, stay brackets. + * + * The link targets ``#cite-N``, which `MarkdownAnswer` renders as the "jump to + * source" pill. + */ +import type { PhrasingContent, Root, Text } from 'mdast'; +import { asciiPunctuation } from 'micromark-util-character'; +import { SKIP, visit } from 'unist-util-visit'; + +export type CitationOptions = { + /** + * How many sources the answer has; only ``[1]`` to ``[sourceCount]`` are + * linked, since any other number has no source to jump to. Unset links + * every ``[N]``. + */ + sourceCount?: number; +}; + +const CITATION = /\[(\d+)\]/g; + +// The offsets in `node.value` of characters the source wrote as a backslash +// escape. The parser resolves escapes, so ``\[1\]`` and ``[1]`` are the same +// text; only the source tells them apart. Anything else that makes the value +// differ from the source (an entity, a quote marker) ends the walk, and the +// rest of the node counts as unescaped. +function escapedOffsets(node: Text, source: string): Set { + const escaped = new Set(); + const start = node.position?.start.offset; + const end = node.position?.end.offset; + if (start === undefined || end === undefined) return escaped; + const raw = source.slice(start, end); + if (!raw.includes('\\')) return escaped; + + const { value } = node; + let i = 0; + for (let j = 0; j < value.length && i < raw.length; j++) { + // Indentation the parser dropped from a continuation line. + while (raw[i] !== value[j] && (raw[i] === ' ' || raw[i] === '\t')) i++; + if ( + raw[i] === '\\' && + raw[i + 1] === value[j] && + asciiPunctuation(value.charCodeAt(j)) + ) { + escaped.add(j); + i += 2; + } else if (raw[i] === value[j]) { + i++; + } else { + break; + } + } + return escaped; +} + +export function remarkCitations({ sourceCount }: CitationOptions = {}) { + const citable = (n: number) => + sourceCount === undefined || (n >= 1 && n <= sourceCount); + + return (tree: Root, file: { value?: unknown }) => { + const source = String(file.value ?? ''); + + visit(tree, (node, index, parent) => { + if (node.type === 'link' || node.type === 'linkReference') return SKIP; + if (node.type !== 'text' || !parent || index === undefined) return; + + const escaped = escapedOffsets(node, source); + const parts: PhrasingContent[] = []; + let last = 0; + for (const match of node.value.matchAll(CITATION)) { + if (!citable(Number(match[1])) || escaped.has(match.index)) continue; + if (match.index > last) { + parts.push({ + type: 'text', + value: node.value.slice(last, match.index), + }); + } + parts.push({ + type: 'link', + url: `#cite-${match[1]}`, + children: [{ type: 'text', value: match[1] }], + }); + last = match.index + match[0].length; + } + if (parts.length === 0) return; + if (last < node.value.length) { + parts.push({ type: 'text', value: node.value.slice(last) }); + } + parent.children.splice(index, 1, ...parts); + return [SKIP, index + parts.length]; + }); + }; +} diff --git a/frontend/src/conversation/markdown/remarkDisplayMath.ts b/frontend/src/conversation/markdown/remarkDisplayMath.ts new file mode 100644 index 00000000..5de372af --- /dev/null +++ b/frontend/src/conversation/markdown/remarkDisplayMath.ts @@ -0,0 +1,134 @@ +/** + * Decides display versus inline math on the syntax tree, where the parser has + * already told math, code and prose apart. + * + * An inline formula becomes display math when + * - it is written with a block delimiter (``$$`` or ``\[``) and sits on a line + * of its own: ``$$E=mc^2$$`` alone in a paragraph, or a ``\[ … \]`` line the + * flow construct did not take (a lazy line inside a quote, say); or + * - KaTeX can only typeset it in display mode (``\tag``, or an environment such + * as ``align``); inline, it would be a red error instead of an equation. + * + * A promoted formula splits its paragraph: the text before and after stays a + * paragraph of its own. Where no paragraph can be split (a table cell, a + * heading) a display-only formula is still typeset in display mode, in place. + */ +import type { Paragraph, PhrasingContent, Root, RootContent } from 'mdast'; +import { SKIP, visit } from 'unist-util-visit'; + +type InlineMath = Extract; + +// `\tag` and the environments KaTeX refuses outside display mode. +const DISPLAY_ONLY = + /\\tag\b|\\begin\{(?:equation|align|alignat|gather|split|CD)\*?\}/; + +const displayClasses = ['language-math', 'math-display']; + +function isInlineMath(node: PhrasingContent): node is InlineMath { + return node.type === 'inlineMath'; +} + +function blockDelimited(node: InlineMath, source: string): boolean { + const start = node.position?.start.offset; + if (start === undefined) return false; + const opener = source.slice(start, start + 2); + return opener === '$$' || opener === '\\['; +} + +// Whether the siblings either side of `index` end and start a line. +function aloneOnLine(children: PhrasingContent[], index: number): boolean { + const before = children[index - 1]; + const after = children[index + 1]; + const endsLine = + !before || + before.type === 'break' || + (before.type === 'text' && /(?:^|\n)[ \t]*$/.test(before.value)); + const startsLine = + !after || + after.type === 'break' || + (after.type === 'text' && /^[ \t]*(?:\n|$)/.test(after.value)); + return endsLine && startsLine; +} + +function toDisplay(node: InlineMath): RootContent { + return { + type: 'math', + meta: null, + value: node.value, + position: node.position, + data: { + hName: 'pre', + hChildren: [ + { + type: 'element', + tagName: 'code', + properties: { className: displayClasses }, + children: [{ type: 'text', value: node.value }], + }, + ], + }, + }; +} + +// The paragraph around each promoted formula, cut at it, with the line breaks +// that separated them dropped. +function splitParagraph( + paragraph: Paragraph, + promote: (child: PhrasingContent, index: number) => boolean, +): RootContent[] { + const blocks: RootContent[] = []; + let run: PhrasingContent[] = []; + let trimNext = false; + + const flush = () => { + const last = run[run.length - 1]; + if (last?.type === 'break') run.pop(); + else if (last?.type === 'text') last.value = last.value.trimEnd(); + run = run.filter((child) => child.type !== 'text' || child.value !== ''); + if (run.length > 0) blocks.push({ type: 'paragraph', children: run }); + run = []; + }; + + paragraph.children.forEach((child, index) => { + if (isInlineMath(child) && promote(child, index)) { + flush(); + blocks.push(toDisplay(child)); + trimNext = true; + return; + } + if (trimNext) { + trimNext = false; + if (child.type === 'break') return; + if (child.type === 'text') child.value = child.value.trimStart(); + } + run.push(child); + }); + flush(); + return blocks; +} + +export function remarkDisplayMath() { + return (tree: Root, file: { value?: unknown }) => { + const source = String(file.value); + + visit(tree, 'paragraph', (node, index, parent) => { + if (!parent || index === undefined) return; + const promote = (child: PhrasingContent, at: number) => + isInlineMath(child) && + (DISPLAY_ONLY.test(child.value) || + (blockDelimited(child, source) && aloneOnLine(node.children, at))); + if (!node.children.some(promote)) return; + const blocks = splitParagraph(node, promote); + // A paragraph's parent (the root, a list item, a quote) holds blocks. + (parent.children as RootContent[]).splice(index, 1, ...blocks); + return [SKIP, index + blocks.length]; + }); + + // Nowhere to split (a table cell, a heading, inside bold): typeset a + // display-only formula in display mode where it stands. + visit(tree, 'inlineMath', (node) => { + if (!DISPLAY_ONLY.test(node.value)) return; + node.data = { ...node.data, hProperties: { className: displayClasses } }; + }); + }; +} diff --git a/frontend/src/conversation/markdown/singleDollarMath.ts b/frontend/src/conversation/markdown/singleDollarMath.ts new file mode 100644 index 00000000..7d2a96f4 --- /dev/null +++ b/frontend/src/conversation/markdown/singleDollarMath.ts @@ -0,0 +1,160 @@ +/** + * Single-dollar inline math, ``$x$``, that leaves prices alone. + * + * Ported from LibreChat's `client/src/utils/latex.ts` + * (https://github.com/danny-avila/LibreChat), MIT License, + * Copyright (c) 2026 LibreChat. + * + * `$...$` is ambiguous with prices ("from $2bn to at least $4bn"), and only + * the tokenizer can decide a span without rewriting the message: code spans, + * fences and autolinks are structurally excluded, and a rejected span stays + * byte-identical text. `remark-math` keeps running with + * `singleDollarTextMath: false`; this construct is the only single-dollar path. + * + * A `$...$` span becomes math only when all of Pandoc's boundary rules hold: + * - the opening `$` is immediately followed by a non-space character; + * - the closing `$` is immediately preceded by a non-space character and not + * immediately followed by a digit (rejects ranges like "$100-$200"); + * - the span stays on one line, contains no backtick, treats `\`-pairs as + * opaque (`\$` stays inside the span), and closes with balanced braces. + * + * A failed close abandons the whole attempt instead of scanning further, so + * the next `$` in "Price is $50 and $100" can never silently extend a span; + * micromark then retries the construct at that `$` on its own merits. Pandoc + * has the same known false positive: `$HOME/$USER` in prose (not code) is math. + */ +import { + asciiDigit, + markdownLineEnding, + markdownSpace, +} from 'micromark-util-character'; +import { codes, types } from 'micromark-util-symbol'; +import type { + Code, + Construct, + Effects, + Extension, + State, + TokenizeContext, +} from 'micromark-util-types'; +import type { Processor } from 'unified'; + +import { addMicromarkExtension } from './micromarkExtension'; + +function tokenizeMathSpan( + this: TokenizeContext, + effects: Effects, + ok: State, + nok: State, +): State { + let previousCode: Code = null; + let braceDepth = 0; + + return start; + + function start(code: Code): State | undefined { + effects.enter('mathText'); + effects.enter('mathTextSequence'); + effects.consume(code); + effects.exit('mathTextSequence'); + return open; + } + + function open(code: Code): State | undefined { + if ( + code === codes.eof || + code === codes.dollarSign || + code === codes.graveAccent || + markdownSpace(code) || + markdownLineEnding(code) + ) { + return nok(code); + } + effects.enter('mathTextData'); + return content(code); + } + + function content(code: Code): State | undefined { + if ( + code === codes.eof || + code === codes.graveAccent || + markdownLineEnding(code) + ) { + return nok(code); + } + if (code === codes.dollarSign) { + if (markdownSpace(previousCode) || braceDepth !== 0) { + return nok(code); + } + effects.exit('mathTextData'); + effects.enter('mathTextSequence'); + effects.consume(code); + return close; + } + if (code === codes.backslash) { + effects.consume(code); + return escape; + } + if (code === codes.leftCurlyBrace) { + braceDepth++; + } + if (code === codes.rightCurlyBrace) { + if (braceDepth === 0) { + return nok(code); + } + braceDepth--; + } + previousCode = code; + effects.consume(code); + return content; + } + + function escape(code: Code): State | undefined { + if (code === codes.eof || markdownLineEnding(code)) { + return nok(code); + } + previousCode = code; + effects.consume(code); + return content; + } + + function close(code: Code): State | undefined { + if (code === codes.dollarSign || asciiDigit(code)) { + return nok(code); + } + effects.exit('mathTextSequence'); + effects.exit('mathText'); + return ok(code); + } +} + +/** Mirrors `micromark-extension-math`: a `$` opener is valid unless it follows an unescaped `$`. */ +function previous(this: TokenizeContext, code: Code): boolean { + return ( + code !== codes.dollarSign || + this.events[this.events.length - 1][1].type === types.characterEscape + ); +} + +const mathSpan: Construct = { + name: 'mathSpanSingleDollar', + tokenize: tokenizeMathSpan, + previous, +}; + +/** + * micromark syntax extension adding currency-safe single-dollar inline math. + * It emits the same `mathText`/`mathTextData` tokens as + * `micromark-extension-math`, so `remark-math`'s `mathFromMarkdown` handlers + * turn its spans into regular `inlineMath` nodes. Registration order relative + * to `remark-math` is immaterial: this construct rejects `$$` openers and + * `remark-math` (with `singleDollarTextMath: false`) rejects single `$` openers. + */ +export const singleDollarMath: Extension = { + text: { [codes.dollarSign]: mathSpan }, +}; + +/** remark plugin for {@link singleDollarMath}; `remark-math` must run alongside it. */ +export function remarkSingleDollarMath(this: Processor) { + addMicromarkExtension(this, singleDollarMath); +} diff --git a/frontend/src/conversation/markdown/texMath.ts b/frontend/src/conversation/markdown/texMath.ts new file mode 100644 index 00000000..6d27f0d8 --- /dev/null +++ b/frontend/src/conversation/markdown/texMath.ts @@ -0,0 +1,398 @@ +/** + * LaTeX's own math delimiters, ``\( … \)`` and ``\[ … \]``, as micromark + * constructs. Models write math this way far more often than with dollars, and + * remark-math only knows dollars. + * + * The constructs emit the same tokens as `micromark-extension-math`, so + * `remark-math`'s `mathFromMarkdown` turns them into ordinary `inlineMath` and + * `math` nodes and rehype-katex renders them. `remark-math` has to run too. + * + * - Text: ``\(x\)`` and ``\[x\]`` anywhere in a paragraph become inline math. + * The span may wrap lines but not cross a paragraph; ``\\`` pairs are opaque + * (a LaTeX line break), a nested ``\( … \)`` of the same kind is kept whole, + * and whitespace-only content is not math, so an escaped task box + * ``- \[ \] todo`` stays text. ``\[`` right after a letter or digit is an + * escaped index (``arr\[0\]``), not math. + * - Flow: a ``\[`` that starts a line opens display math, closed by a ``\]`` + * that ends a line (the same line or a later one). Its lines are not parsed + * as markdown, so a lone ``=`` is not a setext underline. If a ``\]`` is + * followed by more text on its line, or the block is never closed, it is not + * a display block, and the text construct gets the ``\[`` instead: one line + * such as ``\[a\] and more`` must not swallow the rest of the answer. + * + * Written for this app rather than aliasing `micromark-extension-llm-math` + * over `micromark-extension-math`, whose flow construct does exactly that + * swallowing. Its structure follows that package and `micromark-extension-math` + * (both MIT). + */ +import { + asciiAlphanumeric, + markdownLineEnding, + markdownSpace, +} from 'micromark-util-character'; +import { codes, types } from 'micromark-util-symbol'; +import type { + Code, + Construct, + Effects, + Extension, + State, + TokenizeContext, +} from 'micromark-util-types'; +import type { Processor } from 'unified'; + +import { addMicromarkExtension } from './micromarkExtension'; + +function tokenizeTexText( + this: TokenizeContext, + effects: Effects, + ok: State, + nok: State, +): State { + // What precedes the opening backslash, read before it is consumed. + const before = this.previous; + let open: Code = null; + let close: Code = null; + let depth = 0; + let hasContent = false; + + const closing: Construct = { tokenize: tokenizeClosing, partial: true }; + + return start; + + function start(code: Code): State | undefined { + effects.enter('mathText'); + effects.enter('mathTextSequence'); + effects.consume(code); + return sequenceOpen; + } + + function sequenceOpen(code: Code): State | undefined { + if (code === codes.leftParenthesis) { + close = codes.rightParenthesis; + } else if (code === codes.leftSquareBracket && !asciiAlphanumeric(before)) { + close = codes.rightSquareBracket; + } else { + return nok(code); + } + open = code; + effects.consume(code); + effects.exit('mathTextSequence'); + return between; + } + + function between(code: Code): State | undefined { + if (code === codes.eof) return nok(code); + if (code === codes.backslash) { + return effects.check(closing, closeSequence, escape)(code); + } + // A line ending is kept as data: TeX reads it as a space, and dropping it + // would fuse ``\alpha`` with the next line's first letter. + if (markdownLineEnding(code)) { + effects.enter('mathTextData'); + effects.consume(code); + effects.exit('mathTextData'); + return between; + } + effects.enter('mathTextData'); + return data(code); + } + + function data(code: Code): State | undefined { + if ( + code === codes.eof || + code === codes.backslash || + markdownLineEnding(code) + ) { + effects.exit('mathTextData'); + return between(code); + } + if (!markdownSpace(code)) hasContent = true; + effects.consume(code); + return data; + } + + function escape(code: Code): State | undefined { + effects.enter('mathTextData'); + effects.consume(code); + hasContent = true; + return escaped; + } + + function escaped(code: Code): State | undefined { + if (code === codes.eof || markdownLineEnding(code)) { + effects.exit('mathTextData'); + return between(code); + } + if (code === open) depth++; + // Only reached for a nested close: the outermost one is `closing`. + if (code === close) depth--; + effects.consume(code); + return data; + } + + function closeSequence(code: Code): State | undefined { + if (!hasContent) return nok(code); + effects.enter('mathTextSequence'); + effects.consume(code); + return closeBracket; + } + + function closeBracket(code: Code): State | undefined { + effects.consume(code); + return done; + } + + function done(code: Code): State | undefined { + effects.exit('mathTextSequence'); + effects.exit('mathText'); + return ok(code); + } + + function tokenizeClosing(effects: Effects, ok: State, nok: State): State { + return (code) => { + effects.enter('mathTextSequence'); + effects.consume(code); + return (next) => { + if (next !== close || depth !== 0) return nok(next); + effects.exit('mathTextSequence'); + return ok(next); + }; + }; + } +} + +/** A ``\(`` or ``\[`` opens math unless its backslash is itself escaped. */ +function previousTex(this: TokenizeContext, code: Code): boolean { + return ( + code !== codes.backslash || + this.events[this.events.length - 1][1].type === types.characterEscape + ); +} + +const texText: Construct = { + name: 'mathTexText', + tokenize: tokenizeTexText, + previous: previousTex, +}; + +function tokenizeTexFlow( + this: TokenizeContext, + effects: Effects, + ok: State, + nok: State, +): State { + // Set while micromark asks whether the block may interrupt a paragraph. + const interrupt = this.interrupt; + // Whether the line being entered is a lazy continuation. + const lazyLine = () => this.parser.lazy[this.now().line]; + const tail = this.events[this.events.length - 1]; + // The fence's own indent, stripped from the lines inside it too. + const initialSize = + tail && tail[1].type === types.linePrefix + ? tail[2].sliceSerialize(tail[1], true).length + : 0; + + const closingFence: Construct = { + tokenize: tokenizeClosingFence, + partial: true, + }; + const nonLazyLine: Construct = { + tokenize: tokenizeNonLazyLine, + partial: true, + }; + + return start; + + function start(code: Code): State | undefined { + effects.enter('mathFlow'); + effects.enter('mathFlowFence'); + effects.enter('mathFlowFenceSequence'); + effects.consume(code); + return sequenceOpen; + } + + function sequenceOpen(code: Code): State | undefined { + if (code !== codes.leftSquareBracket) return nok(code); + effects.consume(code); + effects.exit('mathFlowFenceSequence'); + return fenceWhitespace; + } + + function fenceWhitespace(code: Code): State | undefined { + if (!markdownSpace(code)) return fenceEnd(code); + effects.enter(types.whitespace); + return fenceWhitespaceInside(code); + } + + function fenceWhitespaceInside(code: Code): State | undefined { + if (markdownSpace(code)) { + effects.consume(code); + return fenceWhitespaceInside; + } + effects.exit(types.whitespace); + return fenceEnd(code); + } + + function fenceEnd(code: Code): State | undefined { + effects.exit('mathFlowFence'); + return lineContent(code); + } + + // Anywhere in a line of the block, outside a value chunk. + function lineContent(code: Code): State | undefined { + if (code === codes.eof) return nok(code); + if (markdownLineEnding(code)) { + // To interrupt a paragraph only the opening line has to hold up. + if (interrupt) return ok(code); + return effects.attempt(nonLazyLine, lineStart, nok)(code); + } + if (code === codes.backslash) { + return effects.check( + { tokenize: tokenizeBracketClose, partial: true }, + atClose, + valueEscape, + )(code); + } + effects.enter('mathFlowValue'); + return value(code); + } + + function lineStart(code: Code): State | undefined { + if (initialSize === 0 || !markdownSpace(code)) return lineContent(code); + effects.enter(types.linePrefix); + return linePrefix(code, 0); + } + + function linePrefix(code: Code, size: number): State | undefined { + if (markdownSpace(code) && size < initialSize) { + effects.consume(code); + return (next: Code) => linePrefix(next, size + 1); + } + effects.exit(types.linePrefix); + return lineContent(code); + } + + function value(code: Code): State | undefined { + if ( + code === codes.eof || + code === codes.backslash || + markdownLineEnding(code) + ) { + effects.exit('mathFlowValue'); + return lineContent(code); + } + effects.consume(code); + return value; + } + + // A ``\]`` closes the block only when the rest of its line is blank; with + // anything after it, this is a line of prose, not a display block. + function atClose(code: Code): State | undefined { + return effects.attempt(closingFence, after, nok)(code); + } + + function after(code: Code): State | undefined { + effects.exit('mathFlow'); + return ok(code); + } + + // ``\\`` and ``\x`` pairs are content, so ``\\]`` is not a close. + function valueEscape(code: Code): State | undefined { + effects.enter('mathFlowValue'); + effects.consume(code); + return valueEscaped; + } + + function valueEscaped(code: Code): State | undefined { + if (code === codes.eof || markdownLineEnding(code)) { + effects.exit('mathFlowValue'); + return lineContent(code); + } + effects.consume(code); + return value; + } + + function tokenizeBracketClose( + effects: Effects, + ok: State, + nok: State, + ): State { + return (code) => { + effects.enter('mathFlowFenceSequence'); + effects.consume(code); + return (next) => { + if (next !== codes.rightSquareBracket) return nok(next); + effects.exit('mathFlowFenceSequence'); + return ok(next); + }; + }; + } + + function tokenizeClosingFence( + effects: Effects, + ok: State, + nok: State, + ): State { + return (code) => { + effects.enter('mathFlowFence'); + effects.enter('mathFlowFenceSequence'); + effects.consume(code); + return (bracket) => { + effects.consume(bracket); + effects.exit('mathFlowFenceSequence'); + return afterSequence; + }; + }; + + function afterSequence(code: Code): State | undefined { + if (markdownSpace(code)) { + effects.enter(types.whitespace); + return trailing(code); + } + return end(code); + } + + function trailing(code: Code): State | undefined { + if (markdownSpace(code)) { + effects.consume(code); + return trailing; + } + effects.exit(types.whitespace); + return end(code); + } + + function end(code: Code): State | undefined { + if (code !== codes.eof && !markdownLineEnding(code)) return nok(code); + effects.exit('mathFlowFence'); + return ok(code); + } + } + + // Past a line ending, unless the next line is a lazy continuation, which + // means the container around the block (a quote, a list item) has ended. + function tokenizeNonLazyLine(effects: Effects, ok: State, nok: State): State { + return (code) => { + effects.enter(types.lineEnding); + effects.consume(code); + effects.exit(types.lineEnding); + return (next) => (lazyLine() ? nok(next) : ok(next)); + }; + } +} + +const texFlow: Construct = { + name: 'mathTexFlow', + tokenize: tokenizeTexFlow, + concrete: true, +}; + +export const texMath: Extension = { + flow: { [codes.backslash]: texFlow }, + text: { [codes.backslash]: texText }, +}; + +/** remark plugin for {@link texMath}; `remark-math` must run alongside it. */ +export function remarkTexMath(this: Processor) { + addMicromarkExtension(this, texMath); +} From 845e39969ead44799c3026fdf305b2f7aeca813e Mon Sep 17 00:00:00 2001 From: arc53-machine <232052973+arc53-machine@users.noreply.github.com> Date: Mon, 28 Sep 2026 13:21:28 +0100 Subject: [PATCH 2/4] Link only citations that have a source card to jump to A pill for [N] beyond the answer's sources, or on an answer whose sources are hidden, scrolled to nothing. --- frontend/src/conversation/AnswerFlow.tsx | 4 ++++ frontend/src/conversation/ConversationBubble.tsx | 14 ++++++++++---- 2 files changed, 14 insertions(+), 4 deletions(-) diff --git a/frontend/src/conversation/AnswerFlow.tsx b/frontend/src/conversation/AnswerFlow.tsx index 03d76a81..a7fb4ce4 100644 --- a/frontend/src/conversation/AnswerFlow.tsx +++ b/frontend/src/conversation/AnswerFlow.tsx @@ -26,6 +26,8 @@ type AnswerFlowProps = { // Absent on reload, where the order is synthesized from the flat fields. segments?: AnswerSegment[]; isStreaming?: boolean; + /** How many sources the answer can cite; see `MarkdownAnswer`. */ + sourceCount?: number; agentId?: string; /** Set when the bubble already carries its own progress UI (a research run). */ suppressStatusLine?: boolean; @@ -51,6 +53,7 @@ export default function AnswerFlow({ toolCalls, segments, isStreaming, + sourceCount, agentId, suppressStatusLine, artifacts, @@ -136,6 +139,7 @@ export default function AnswerFlow({ ); } else { + const showSources = !( + DisableSourceFE || + type === 'ERROR' || + sources?.length === 0 || + sources?.some((source) => source.link === 'None') + ); bubble = (
- {DisableSourceFE || - type === 'ERROR' || - sources?.length === 0 || - sources?.some((source) => source.link === 'None') + {!showSources ? null : sources && ( // Stretched, not shrink-to-fit: the grid below sizes off this box, @@ -446,6 +449,9 @@ const ConversationBubble = forwardRef< toolCalls={toolCalls} segments={segments} isStreaming={isStreaming} + // A citation pill jumps to a source card, so an answer whose + // sources are not shown has nothing to cite. + sourceCount={showSources ? (sources?.length ?? 0) : 0} agentId={agentId} // A research run already narrates itself above; the status line // would be a second live indicator away from the point of action. From c0c587de81b7d5381c159fd492da82dd2afaf8f2 Mon Sep 17 00:00:00 2001 From: arc53-machine <232052973+arc53-machine@users.noreply.github.com> Date: Mon, 28 Sep 2026 13:21:28 +0100 Subject: [PATCH 3/4] Keep answer math in its frame and its case A display formula wider than the column scrolls in its own box, as code and tables do, and a formula in a table header is not uppercased: n and N are different variables. --- .../conversation/ConversationBubble.module.css | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/frontend/src/conversation/ConversationBubble.module.css b/frontend/src/conversation/ConversationBubble.module.css index 0fce5a83..b32a1dae 100644 --- a/frontend/src/conversation/ConversationBubble.module.css +++ b/frontend/src/conversation/ConversationBubble.module.css @@ -9,3 +9,18 @@ .list li > .list { margin-top: 0.5em; } + +/* A display formula wider than the answer column scrolls in its own box, as + fenced code and tables do, rather than pushing the column sideways. The + padding keeps tall symbols clear of the clipped edge. */ +.answer :global(.katex-display) { + overflow-x: auto; + overflow-y: hidden; + padding-block: 0.25em; +} + +/* Table headers are set in capitals, but a formula keeps its case: `n` and + `N` are different variables. */ +.answer thead :global(.katex) { + text-transform: none; +} From 1c9cf7b6932ab77df0becd3daf4cab3f37eda13c Mon Sep 17 00:00:00 2001 From: arc53-machine <232052973+arc53-machine@users.noreply.github.com> Date: Mon, 28 Sep 2026 13:30:55 +0100 Subject: [PATCH 4/4] Open the sources list from a citation that has no card Only the first three sources are shown as cards, so a pill for [4] or later scrolled to nothing; it now opens the full list. --- frontend/src/conversation/AnswerFlow.tsx | 3 ++ .../src/conversation/ConversationBubble.tsx | 11 +++++- .../conversation/MarkdownAnswer.math.test.tsx | 39 ++++++++++++++++++- frontend/src/conversation/MarkdownAnswer.tsx | 8 +++- 4 files changed, 58 insertions(+), 3 deletions(-) diff --git a/frontend/src/conversation/AnswerFlow.tsx b/frontend/src/conversation/AnswerFlow.tsx index a7fb4ce4..9e134617 100644 --- a/frontend/src/conversation/AnswerFlow.tsx +++ b/frontend/src/conversation/AnswerFlow.tsx @@ -28,6 +28,7 @@ type AnswerFlowProps = { isStreaming?: boolean; /** How many sources the answer can cite; see `MarkdownAnswer`. */ sourceCount?: number; + onOpenSources?: () => void; agentId?: string; /** Set when the bubble already carries its own progress UI (a research run). */ suppressStatusLine?: boolean; @@ -54,6 +55,7 @@ export default function AnswerFlow({ segments, isStreaming, sourceCount, + onOpenSources, agentId, suppressStatusLine, artifacts, @@ -140,6 +142,7 @@ export default function AnswerFlow({ content={message} isStreaming={isStreaming} sourceCount={sourceCount} + onOpenSources={onOpenSources} artifacts={artifacts} turnArtifacts={turnArtifacts} onOpenArtifact={onOpenArtifact} diff --git a/frontend/src/conversation/ConversationBubble.tsx b/frontend/src/conversation/ConversationBubble.tsx index cd5234d5..fba3e819 100644 --- a/frontend/src/conversation/ConversationBubble.tsx +++ b/frontend/src/conversation/ConversationBubble.tsx @@ -16,7 +16,14 @@ import { ThumbsUp, ExternalLink, } from 'lucide-react'; -import { forwardRef, Fragment, useEffect, useRef, useState } from 'react'; +import { + forwardRef, + Fragment, + useCallback, + useEffect, + useRef, + useState, +} from 'react'; import { useTranslation } from 'react-i18next'; import { useSelector } from 'react-redux'; @@ -126,6 +133,7 @@ const ConversationBubble = forwardRef< const [shouldShowToggle, setShouldShowToggle] = useState(false); const [isSidebarOpen, setIsSidebarOpen] = useState(false); + const openSources = useCallback(() => setIsSidebarOpen(true), []); const editableQueryRef = useRef(null); const [isQuestionCollapsed, setIsQuestionCollapsed] = useState(true); @@ -452,6 +460,7 @@ const ConversationBubble = forwardRef< // A citation pill jumps to a source card, so an answer whose // sources are not shown has nothing to cite. sourceCount={showSources ? (sources?.length ?? 0) : 0} + onOpenSources={openSources} agentId={agentId} // A research run already narrates itself above; the status line // would be a second live indicator away from the point of action. diff --git a/frontend/src/conversation/MarkdownAnswer.math.test.tsx b/frontend/src/conversation/MarkdownAnswer.math.test.tsx index c52cf656..63d3d689 100644 --- a/frontend/src/conversation/MarkdownAnswer.math.test.tsx +++ b/frontend/src/conversation/MarkdownAnswer.math.test.tsx @@ -51,7 +51,11 @@ afterEach(() => { container.remove(); }); -type RenderOptions = { isStreaming?: boolean; sourceCount?: number }; +type RenderOptions = { + isStreaming?: boolean; + sourceCount?: number; + onOpenSources?: () => void; +}; function render(content: string, options: RenderOptions = {}) { act(() => root.render()); @@ -418,6 +422,39 @@ describe('citations', () => { }); }); +describe('citation pills', () => { + function click(label: string) { + const pill = Array.from(container.querySelectorAll('button')).find( + (node) => node.textContent === label, + ); + act(() => pill?.click()); + } + + it('scrolls to the source card when it is shown', () => { + const card = document.createElement('div'); + card.id = 'source-1'; + card.scrollIntoView = vi.fn(); + document.body.appendChild(card); + const onOpenSources = vi.fn(); + try { + render('Per [2].', { sourceCount: 5, onOpenSources }); + click('2'); + expect(card.scrollIntoView).toHaveBeenCalled(); + expect(onOpenSources).not.toHaveBeenCalled(); + } finally { + card.remove(); + } + }); + + it('opens the sources list for a source with no card', () => { + // Only the first three sources get a card; `[5]` has none to scroll to. + const onOpenSources = vi.fn(); + render('Per [5].', { sourceCount: 5, onOpenSources }); + click('5'); + expect(onOpenSources).toHaveBeenCalledTimes(1); + }); +}); + describe('mermaid', () => { it('renders a closed mermaid fence as a diagram, source untouched', () => { render( diff --git a/frontend/src/conversation/MarkdownAnswer.tsx b/frontend/src/conversation/MarkdownAnswer.tsx index 39d4bedb..a1cfa469 100644 --- a/frontend/src/conversation/MarkdownAnswer.tsx +++ b/frontend/src/conversation/MarkdownAnswer.tsx @@ -83,6 +83,7 @@ export default function MarkdownAnswer({ artifacts: artifactsProp, turnArtifacts: turnArtifactsProp, onOpenArtifact, + onOpenSources, }: { content: string; isStreaming?: boolean; @@ -100,6 +101,8 @@ export default function MarkdownAnswer({ /** This turn's own artifacts; the filename fallback prefers them. */ turnArtifacts?: SandboxArtifact[]; onOpenArtifact?: (artifact: { id: string; toolName: string }) => void; + /** Opens the full sources list, for a citation whose card is not shown. */ + onOpenSources?: () => void; }) { const { t } = useTranslation(); const [isDarkTheme] = useDarkTheme(); @@ -194,6 +197,9 @@ export default function MarkdownAnswer({ () => el.classList.remove('ring-3', 'ring-primary'), 2000, ); + } else { + // Only the first few sources get a card to scroll to. + onOpenSources?.(); } }} className="mx-0.5 h-5 min-w-5" @@ -321,7 +327,7 @@ export default function MarkdownAnswer({ return {children}; }, }; - }, [t, isDarkTheme, artifacts, turnArtifacts, onOpenArtifact]); + }, [t, isDarkTheme, artifacts, turnArtifacts, onOpenArtifact, onOpenSources]); return ( <>