Merge pull request #2840 from arc53-machine/math-rendering-parser-level

Render chat math in the markdown parser: \( \), \[ \], $x$, mhchem, streaming
This commit is contained in:
Alex authored and GitHub committed 2026-09-28 13:50:39 +01:00
commit 1c19ed8ce2
15 files changed
+2191 -520

No files matched your search

+18 -4
View File
@@ -21,6 +21,8 @@
"lodash": "^4.18.1",
"lucide-react": "^1.45.0",
"mermaid": "^12.0.0",
"micromark-util-character": "^2.1.1",
"micromark-util-symbol": "^2.0.1",
"radix-ui": "^1.6.7",
"react": "^19.3.0",
"react-chartjs-2": "^5.3.1",
@@ -38,7 +40,11 @@
"rehype-katex": "^7.0.1",
"remark-gfm": "^4.0.1",
"remark-math": "^6.0.0",
"tailwind-merge": "^3.6.0"
"remark-parse": "^11.0.0",
"remend": "^1.3.1",
"tailwind-merge": "^3.6.0",
"unified": "^11.0.5",
"unist-util-visit": "^5.1.0"
},
"devDependencies": {
"@eslint/js": "^9.39.5",
@@ -46,6 +52,7 @@
"@tailwindcss/postcss": "^4.3.3",
"@types/d3-force": "^3.0.10",
"@types/lodash": "^4.17.25",
"@types/mdast": "^4.0.4",
"@types/react": "^19.3.0",
"@types/react-dom": "^19.3.0",
"@types/react-syntax-highlighter": "^15.5.13",
@@ -64,6 +71,7 @@
"happy-dom": "^20.14.5",
"husky": "^9.1.7",
"lint-staged": "^17.5.1",
"micromark-util-types": "^2.0.3",
"postcss": "^8.5.28",
"prettier": "^3.9.6",
"prettier-plugin-tailwindcss": "^0.8.1",
@@ -9911,9 +9919,9 @@
"license": "MIT"
},
"node_modules/micromark-util-types": {
"version": "2.0.2",
"resolved": "https://registry.npmjs.org/micromark-util-types/-/micromark-util-types-2.0.2.tgz",
"integrity": "sha512-Yw0ECSpJoViF1qTU4DC6NwtC4aWGt1EkzaQB8KPPyCRR8z9TWeV0HbEFGTO+ZY1wB22zmxnJqhPyTpOVCpeHTA==",
"version": "2.0.3",
"resolved": "https://registry.npmjs.org/micromark-util-types/-/micromark-util-types-2.0.3.tgz",
"integrity": "sha512-oxB2Ik03hI0gv+VNn9tnh1t1YEe9MDPptViAEgfdf3YQHsn0pzGTgCdlSCJXcwhqm8phaEuM7zeEu3QQzVBrPg==",
"funding": [
{
"type": "GitHub Sponsors",
@@ -11271,6 +11279,12 @@
"url": "https://opencollective.com/unified"
}
},
"node_modules/remend": {
"version": "1.3.1",
"resolved": "https://registry.npmjs.org/remend/-/remend-1.3.1.tgz",
"integrity": "sha512-N3DiY5qbRPoa5vkxn1oDLMyXOVTeo6Hp+XOj6SIqJAYUgLS0Q587gILPMom/qm86AQ/ZrcOdwEIzCz8V3J0nxQ==",
"license": "Apache-2.0"
},
"node_modules/reselect": {
"version": "5.3.0",
"resolved": "https://registry.npmjs.org/reselect/-/reselect-5.3.0.tgz",
+9 -1
View File
@@ -38,6 +38,8 @@
"lodash": "^4.18.1",
"lucide-react": "^1.45.0",
"mermaid": "^12.0.0",
"micromark-util-character": "^2.1.1",
"micromark-util-symbol": "^2.0.1",
"radix-ui": "^1.6.7",
"react": "^19.3.0",
"react-chartjs-2": "^5.3.1",
@@ -55,7 +57,11 @@
"rehype-katex": "^7.0.1",
"remark-gfm": "^4.0.1",
"remark-math": "^6.0.0",
"tailwind-merge": "^3.6.0"
"remark-parse": "^11.0.0",
"remend": "^1.3.1",
"tailwind-merge": "^3.6.0",
"unified": "^11.0.5",
"unist-util-visit": "^5.1.0"
},
"devDependencies": {
"@eslint/js": "^9.39.5",
@@ -63,6 +69,7 @@
"@tailwindcss/postcss": "^4.3.3",
"@types/d3-force": "^3.0.10",
"@types/lodash": "^4.17.25",
"@types/mdast": "^4.0.4",
"@types/react": "^19.3.0",
"@types/react-dom": "^19.3.0",
"@types/react-syntax-highlighter": "^15.5.13",
@@ -81,6 +88,7 @@
"happy-dom": "^20.14.5",
"husky": "^9.1.7",
"lint-staged": "^17.5.1",
"micromark-util-types": "^2.0.3",
"postcss": "^8.5.28",
"prettier": "^3.9.6",
"prettier-plugin-tailwindcss": "^0.8.1",
+7
View File
@@ -26,6 +26,9 @@ type AnswerFlowProps = {
// Absent on reload, where the order is synthesized from the flat fields.
segments?: AnswerSegment[];
isStreaming?: boolean;
/** How many sources the answer can cite; see `MarkdownAnswer`. */
sourceCount?: number;
onOpenSources?: () => void;
agentId?: string;
/** Set when the bubble already carries its own progress UI (a research run). */
suppressStatusLine?: boolean;
@@ -51,6 +54,8 @@ export default function AnswerFlow({
toolCalls,
segments,
isStreaming,
sourceCount,
onOpenSources,
agentId,
suppressStatusLine,
artifacts,
@@ -136,6 +141,8 @@ export default function AnswerFlow({
<MarkdownAnswer
content={message}
isStreaming={isStreaming}
sourceCount={sourceCount}
onOpenSources={onOpenSources}
artifacts={artifacts}
turnArtifacts={turnArtifacts}
onOpenArtifact={onOpenArtifact}
@@ -9,3 +9,18 @@
.list li > .list {
margin-top: 0.5em;
}
/* A display formula wider than the answer column scrolls in its own box, as
fenced code and tables do, rather than pushing the column sideways. The
padding keeps tall symbols clear of the clipped edge. */
.answer :global(.katex-display) {
overflow-x: auto;
overflow-y: hidden;
padding-block: 0.25em;
}
/* Table headers are set in capitals, but a formula keeps its case: `n` and
`N` are different variables. */
.answer thead :global(.katex) {
text-transform: none;
}
@@ -16,7 +16,14 @@ import {
ThumbsUp,
ExternalLink,
} from 'lucide-react';
import { forwardRef, Fragment, useEffect, useRef, useState } from 'react';
import {
forwardRef,
Fragment,
useCallback,
useEffect,
useRef,
useState,
} from 'react';
import { useTranslation } from 'react-i18next';
import { useSelector } from 'react-redux';
@@ -126,6 +133,7 @@ const ConversationBubble = forwardRef<
const [shouldShowToggle, setShouldShowToggle] = useState(false);
const [isSidebarOpen, setIsSidebarOpen] = useState<boolean>(false);
const openSources = useCallback(() => setIsSidebarOpen(true), []);
const editableQueryRef = useRef<HTMLDivElement>(null);
const [isQuestionCollapsed, setIsQuestionCollapsed] = useState(true);
@@ -279,15 +287,18 @@ const ConversationBubble = forwardRef<
</div>
);
} else {
const showSources = !(
DisableSourceFE ||
type === 'ERROR' ||
sources?.length === 0 ||
sources?.some((source) => source.link === 'None')
);
bubble = (
<div
ref={ref}
className={cn('flex flex-wrap self-start', className, 'group flex-col')}
>
{DisableSourceFE ||
type === 'ERROR' ||
sources?.length === 0 ||
sources?.some((source) => source.link === 'None')
{!showSources
? null
: sources && (
// Stretched, not shrink-to-fit: the grid below sizes off this box,
@@ -446,6 +457,10 @@ const ConversationBubble = forwardRef<
toolCalls={toolCalls}
segments={segments}
isStreaming={isStreaming}
// A citation pill jumps to a source card, so an answer whose
// sources are not shown has nothing to cite.
sourceCount={showSources ? (sources?.length ?? 0) : 0}
onOpenSources={openSources}
agentId={agentId}
// A research run already narrates itself above; the status line
// would be a second live indicator away from the point of action.
@@ -1,194 +0,0 @@
/**
* Code must survive the markdown string pre-passes.
*
* ``processMarkdownContent`` rewrote every ``[N]`` as ``[N](#cite-N)`` and
* ``preprocessLaTeX`` every ``\[…\]`` as ``$$…$$``, neither with any awareness
* of code, so an index subscript or a bash prompt escape came out corrupted in
* the rendered answer — and in what the copy button copied.
*
* Neither rewrite can move into a remark plugin, where a ``code`` node would be
* distinct from a ``text`` one: remark-math tokenises ``$``/``$$`` while
* parsing, so the LaTeX pass has to happen on the raw string. Hence the shared
* ``applyOutsideCode`` masking helper rather than an AST visitor.
*/
import { describe, expect, it } from 'vitest';
import {
applyOutsideCode,
preprocessLaTeX,
processMarkdownContent,
} from './MarkdownAnswer';
/** The rendered text of a single-segment answer. */
const text = (source: string) => processMarkdownContent(source)[0].content;
describe('citation rewriting spares code', () => {
it('leaves an index subscript alone inside a fence', () => {
const source = '```python\nfor row in cur:\n print(row[0])\n```';
expect(text(source)).toBe(source);
});
it('leaves an index subscript alone inside an inline span', () => {
expect(text('Use `items[2]` here.')).toBe('Use `items[2]` here.');
});
it('leaves a mid-sentence triple-backtick span alone', () => {
expect(text('inline ```code[0]``` end')).toBe('inline ```code[0]``` end');
});
it('leaves a triple-backtick span that starts a line alone', () => {
// A backtick fence's info string cannot contain backticks, so this opens a
// code span, not an unclosed fence that swallows the rest of the answer.
expect(text('```code[0]``` and see [1]')).toBe(
'```code[0]``` and see [1](#cite-1)',
);
});
it('closes an inline span only on a run of its own length', () => {
// The `` run does not close the ` span, so `b[1]` is still code.
const source = 'Use `a`` b[1] more` end';
expect(text(source)).toBe(source);
});
it('recognises a fence closed on a CRLF line', () => {
expect(text('```js\r\nconst a = arr[0];\r\n```\r\nSee [1].')).toBe(
'```js\r\nconst a = arr[0];\r\n```\r\nSee [1](#cite-1).',
);
});
it('does not carry an inline span across a blank line', () => {
// The blank line ends the paragraph, so the backticks never pair and
// `bar[1]` is prose — which is how remark parses it too.
expect(text('a `foo\n\nbar[1]` end')).toBe(
'a `foo\n\nbar[1](#cite-1)` end',
);
});
it('leaves an inline span that wraps a line alone', () => {
expect(text('use `arr[0]\nnext` here')).toBe('use `arr[0]\nnext` here');
});
it('leaves a tilde fence alone', () => {
const source = '~~~python\nprint(row[0])\n~~~';
expect(text(source)).toBe(source);
});
it('leaves a fence nested in a longer one alone', () => {
// A ``` run does not close a ```` fence, so the inner block is content.
const source = '````markdown\n```js\nconst a = arr[0];\n```\n````';
expect(text(source)).toBe(source);
});
it('leaves a fence indented under a list item alone', () => {
const source =
'1. Step:\n\n ```js\n const a = arr[0];\n\n const b = arr[1];\n ```';
expect(text(source)).toBe(source);
});
it('leaves an unterminated fence alone while the answer is still streaming', () => {
const source = '```python\nprint(row[0]';
expect(text(source)).toBe(source);
});
it('leaves an unterminated inline span alone while streaming', () => {
expect(text('Use `items[2')).toBe('Use `items[2');
});
it('leaves mermaid node labels alone', () => {
// The cite pass runs before the ```mermaid split, so diagram source was
// corrupted too and the render failed.
const source = '```mermaid\nflowchart TD\n A[0] --> B[1]\n```';
expect(processMarkdownContent(source)[0].content).toBe(
'flowchart TD\n A[0] --> B[1]',
);
});
it('still links a real citation in prose', () => {
expect(text('See source [1] for details.')).toBe(
'See source [1](#cite-1) for details.',
);
});
it('links prose citations on either side of a fence without touching it', () => {
expect(
text('Per [1], run:\n```js\nconst a = arr[0];\n```\nthen [2].'),
).toBe(
'Per [1](#cite-1), run:\n```js\nconst a = arr[0];\n```\nthen [2](#cite-2).',
);
});
it('still links a citation on an indented list continuation line', () => {
// Four-space indents are list content far more often than they are code,
// so they are prose to the masker.
expect(text('- one\n - nested [1]')).toBe(
'- one\n - nested [1](#cite-1)',
);
});
it('leaves an already-linked reference alone', () => {
expect(text('see [1](#cite-1) ok')).toBe('see [1](#cite-1) ok');
});
});
describe('LaTeX preprocessing spares code', () => {
it('leaves bash prompt escapes alone', () => {
// `\[`/`\]` are non-printing-sequence markers in PS1; the block rule turned
// them into `$$` delimiters and broke the prompt.
const source = '```bash\nPS1="\\[\\e[0m\\]$ "\n```';
expect(preprocessLaTeX(source)).toBe(source);
});
it('leaves bracket character classes alone in an inline span', () => {
expect(preprocessLaTeX('match `\\[0-9\\]` here')).toBe(
'match `\\[0-9\\]` here',
);
});
it('still converts real inline math in prose', () => {
expect(preprocessLaTeX('The value \\(x^2\\) is fine.')).toBe(
'The value $x^2$ is fine.',
);
});
it('still converts real block math in prose', () => {
expect(preprocessLaTeX('Given \\[a+b\\] we get')).toBe(
'Given $$a+b$$ we get',
);
});
it('does not link a subscript inside the math it just produced', () => {
// `$a[1](#cite-1)$` is not something KaTeX can render.
expect(text('value \\(a[1]\\) end')).toBe('value $a[1]$ end');
expect(text('given \\[a[1]+b\\] we get [2]')).toBe(
'given $$a[1]+b$$ we get [2](#cite-2)',
);
});
});
describe('applyOutsideCode', () => {
const shout = (segment: string) => segment.toUpperCase();
it('transforms prose either side of every code span', () => {
expect(applyOutsideCode('a `b` c ```d``` e', shout)).toBe(
'A `b` C ```d``` E',
);
});
it('handles several inline spans on one line', () => {
expect(applyOutsideCode('x `a` y `b` z', shout)).toBe('X `a` Y `b` Z');
});
it('treats a double-backtick span containing a backtick as one span', () => {
expect(applyOutsideCode('x ``a`b`` y', shout)).toBe('X ``a`b`` Y');
});
it('does not close a long fence on a shorter run', () => {
expect(applyOutsideCode('````\na\n```\nb\n````\nc', shout)).toBe(
'````\na\n```\nb\n````\nC',
);
});
it('passes an empty string through', () => {
expect(applyOutsideCode('', shout)).toBe('');
});
});
@@ -0,0 +1,209 @@
/**
* Code must survive citation linking and math parsing.
*
* Both used to be regex passes over the raw answer, and both corrupted the
* user's own code: `print(row[0])` rendered as `print(row[0](#cite-0))`, and
* `PS1="\[\e[0m\]"` as `PS1="$$\e[0m$$"`, in the answer and in what the copy
* button copied. Math is now parsed by the markdown parser and citations are
* linked on its syntax tree, where code is never plain text; these cases pin
* that down on the rendered answer.
*/
import { act } from 'react';
import { createRoot, type Root } from 'react-dom/client';
vi.mock('../hooks', () => ({ useDarkTheme: () => [false, vi.fn()] }));
vi.mock('../components/MermaidRenderer', () => ({
default: ({ code }: { code: string }) => (
<div data-testid="mermaid">{code}</div>
),
}));
vi.mock('../components/CopyButton', () => ({ default: () => null }));
import MarkdownAnswer from './MarkdownAnswer';
Object.assign(globalThis, { IS_REACT_ACT_ENVIRONMENT: true });
let container: HTMLDivElement;
let root: Root;
beforeEach(() => {
container = document.createElement('div');
document.body.appendChild(container);
root = createRoot(container);
});
afterEach(() => {
act(() => root.unmount());
container.remove();
});
function render(content: string, isStreaming = false) {
act(() =>
root.render(<MarkdownAnswer content={content} isStreaming={isStreaming} />),
);
}
/** Text of every code block and inline code span, in order. */
function code() {
return Array.from(container.querySelectorAll('code')).map(
(node) => node.textContent,
);
}
/** Citation pills, the only buttons these answers render. */
function citations() {
return Array.from(container.querySelectorAll('button')).map(
(node) => node.textContent,
);
}
function math() {
return container.querySelectorAll('.katex, .katex-error').length;
}
describe('citation linking spares code', () => {
it('leaves an index subscript alone inside a fence', () => {
render('```python\nfor row in cur:\n print(row[0])\n```');
expect(code()).toEqual(['for row in cur:\n print(row[0])']);
expect(citations()).toEqual([]);
});
it('leaves an index subscript alone inside an inline span', () => {
render('Use `items[2]` here.');
expect(code()).toEqual(['items[2]']);
expect(citations()).toEqual([]);
});
it('leaves a mid-sentence triple-backtick span alone', () => {
render('inline ```code[0]``` end');
expect(code()).toEqual(['code[0]']);
expect(citations()).toEqual([]);
});
it('leaves a triple-backtick span that starts a line alone', () => {
// A backtick fence's info string cannot contain backticks, so this opens a
// code span, not an unclosed fence that swallows the rest of the answer.
render('```code[0]``` and see [1]');
expect(code()).toEqual(['code[0]']);
expect(citations()).toEqual(['1']);
});
it('closes an inline span only on a run of its own length', () => {
// The `` run does not close the ` span, so `b[1]` is still code.
render('Use `a`` b[1] more` end');
expect(code()).toEqual(['a`` b[1] more']);
expect(citations()).toEqual([]);
});
it('recognises a fence closed on a CRLF line', () => {
render('```js\r\nconst a = arr[0];\r\n```\r\nSee [1].');
expect(code()).toEqual(['const a = arr[0];']);
expect(citations()).toEqual(['1']);
});
it('does not carry an inline span across a blank line', () => {
// The blank line ends the paragraph, so the backticks never pair and
// `bar[1]` is prose.
render('a `foo\n\nbar[1]` end');
expect(code()).toEqual([]);
expect(citations()).toEqual(['1']);
});
it('leaves an inline span that wraps a line alone', () => {
render('use `arr[0]\nnext` here');
expect(code()).toEqual(['arr[0] next']);
expect(citations()).toEqual([]);
});
it('leaves a tilde fence alone', () => {
render('~~~python\nprint(row[0])\n~~~');
expect(code()).toEqual(['print(row[0])']);
expect(citations()).toEqual([]);
});
it('leaves a fence nested in a longer one alone', () => {
// A ``` run does not close a ```` fence, so the inner block is content.
render('````markdown\n```js\nconst a = arr[0];\n```\n````');
expect(code()).toEqual(['```js\nconst a = arr[0];\n```']);
expect(citations()).toEqual([]);
});
it('leaves a fence indented under a list item alone', () => {
render(
'1. Step:\n\n ```js\n const a = arr[0];\n\n const b = arr[1];\n ```',
);
expect(code()).toEqual(['const a = arr[0];\n\nconst b = arr[1];']);
expect(citations()).toEqual([]);
});
it('leaves an unterminated fence alone while the answer is still streaming', () => {
render('```python\nprint(row[0]', true);
expect(code()).toEqual(['print(row[0]']);
expect(citations()).toEqual([]);
});
it('leaves an unterminated inline span alone while streaming', () => {
render('Use `items[2', true);
expect(code()).toEqual(['items[2']);
expect(citations()).toEqual([]);
});
it('leaves mermaid node labels alone', () => {
render('```mermaid\nflowchart TD\n A[0] --> B[1]\n```');
expect(
container.querySelector('[data-testid="mermaid"]')?.textContent,
).toBe('flowchart TD\n A[0] --> B[1]');
expect(citations()).toEqual([]);
});
it('still links a real citation in prose', () => {
render('See source [1] for details.');
expect(citations()).toEqual(['1']);
expect(container.textContent).toBe('See source 1 for details.');
});
it('links prose citations on either side of a fence without touching it', () => {
render('Per [1], run:\n```js\nconst a = arr[0];\n```\nthen [2].');
expect(code()).toEqual(['const a = arr[0];']);
expect(citations()).toEqual(['1', '2']);
});
it('still links a citation on an indented list continuation line', () => {
render('- one\n - nested [1]');
expect(citations()).toEqual(['1']);
});
it('leaves an already-linked reference alone', () => {
render('see [1](#cite-1) ok');
expect(citations()).toEqual(['1']);
});
});
describe('math parsing spares code', () => {
it('leaves bash prompt escapes alone', () => {
// `\[`/`\]` are non-printing-sequence markers in PS1; the old block rule
// turned them into `$$` delimiters and broke the prompt.
render('```bash\nPS1="\\[\\e[0m\\]$ "\n```');
expect(code()).toEqual(['PS1="\\[\\e[0m\\]$ "']);
expect(math()).toBe(0);
});
it('leaves bracket character classes alone in an inline span', () => {
render('match `\\[0-9\\]` here');
expect(code()).toEqual(['\\[0-9\\]']);
expect(math()).toBe(0);
});
it('leaves dollars alone in an inline span', () => {
render('Run `echo $HOME and $$` then `$x$`.');
expect(code()).toEqual(['echo $HOME and $$', '$x$']);
expect(math()).toBe(0);
});
it('does not link a subscript inside math', () => {
render('value \\(a[1]\\) end, given \\[a[1]+b\\] we get [2]');
expect(math()).toBe(2);
expect(container.querySelectorAll('.katex-error').length).toBe(0);
expect(citations()).toEqual(['2']);
});
});
@@ -0,0 +1,561 @@
/**
* Math must render as math, and only math.
*
* ``processMarkdownContent`` rewrote inline ``\(x\)`` to ``$x$`` while
* remark-math runs with ``singleDollarTextMath: false``, so every inline span
* rendered as the literal text ``$x$`` (students asked "what does $ mean").
* A one-line ``\[ … \]`` became ``$$ … $$`` mid-paragraph, which remark-math
* reads as *inline* math, so display equations shrank into the sentence. And a
* citation pass over the raw string turned ``$$a[1]$$`` into a link inside the
* formula. Math is now parsed by the markdown parser itself; these tests only
* look at what renders.
*/
import { act } from 'react';
import { createRoot, type Root } from 'react-dom/client';
// Records the markdown each ReactMarkdown instance renders.
const markdownRenders = vi.hoisted(() => vi.fn());
vi.mock('react-markdown', async (importOriginal) => {
const actual = await importOriginal<typeof import('react-markdown')>();
return {
...actual,
default: (props: Parameters<typeof actual.default>[0]) => {
markdownRenders(props.children);
return actual.default(props);
},
};
});
vi.mock('../hooks', () => ({ useDarkTheme: () => [false, vi.fn()] }));
vi.mock('../components/MermaidRenderer', () => ({
default: ({ code }: { code: string }) => (
<div data-testid="mermaid">{code}</div>
),
}));
vi.mock('../components/CopyButton', () => ({ default: () => null }));
import MarkdownAnswer from './MarkdownAnswer';
Object.assign(globalThis, { IS_REACT_ACT_ENVIRONMENT: true });
let container: HTMLDivElement;
let root: Root;
beforeEach(() => {
container = document.createElement('div');
document.body.appendChild(container);
root = createRoot(container);
});
afterEach(() => {
act(() => root.unmount());
container.remove();
});
type RenderOptions = {
isStreaming?: boolean;
sourceCount?: number;
onOpenSources?: () => void;
};
function render(content: string, options: RenderOptions = {}) {
act(() => root.render(<MarkdownAnswer content={content} {...options} />));
const all = container.querySelectorAll('.katex').length;
const display = container.querySelectorAll('.katex-display').length;
const clone = container.cloneNode(true) as HTMLElement;
clone
.querySelectorAll('.katex, .katex-error')
.forEach((node) => node.remove());
return {
inline: all - display,
display,
errors: container.querySelectorAll('.katex-error').length,
text: clone.textContent ?? '',
};
}
/** TeX source of every rendered formula, in order. */
function formulas() {
return Array.from(
container.querySelectorAll(
'.katex annotation[encoding="application/x-tex"]',
),
).map((node) => node.textContent);
}
/** Citation pills, the only buttons these answers render. */
function citations() {
return Array.from(container.querySelectorAll('button')).map(
(node) => node.textContent,
);
}
describe('backslash delimiters', () => {
it('renders inline \\( \\) math as inline KaTeX, not a literal $', () => {
const out = render('The velocity is \\(v = u + at\\) at time t.');
expect(out.inline).toBe(1);
expect(out.display).toBe(0);
expect(out.text).not.toContain('$');
expect(out.text).toContain('at time t.');
});
it.each([
['at the start of a line', '\\(v\\) is the final velocity.'],
['in a list item', '- speed \\(v\\) in metres'],
['in a table cell', '| q | f |\n|---|---|\n| v | \\(v=u+at\\) |'],
['wrapping a line', 'where \\(a +\nb\\) holds'],
['in a heading', '## Speed \\(v\\)'],
['inside bold', '**the speed \\(v\\)**'],
])('renders inline math %s', (_, source) => {
const out = render(source);
expect(out.inline).toBe(1);
expect(out.errors).toBe(0);
expect(out.text).not.toContain('$');
});
it('keeps a line break inside inline math as whitespace', () => {
// `\alpha` followed by a newline must not fuse into `\alphab`.
render('then \\(\\alpha\nb\\) holds');
expect(formulas()).toEqual(['\\alpha\nb']);
});
it('renders a dollar sign inside inline math without a KaTeX error', () => {
const out = render('cost \\(C = \\$5 n\\) here');
expect(out.inline).toBe(1);
expect(out.errors).toBe(0);
});
it('renders a one-line \\[ \\] on its own line as display math', () => {
const out = render(
'Displacement:\n\\[ s = ut + \\tfrac12 at^2 \\]\nwhere u is speed.',
);
expect(out.display).toBe(1);
expect(out.inline).toBe(0);
expect(out.text).toContain('Displacement:');
expect(out.text).toContain('where u is speed.');
});
it('renders a one-line \\[ \\] inside a list item as display math', () => {
const out = render(
'1. Step one:\n\n \\[ v^2 = u^2 + 2as \\]\n\n2. Step two',
);
expect(out.display).toBe(1);
expect(container.querySelectorAll('ol > li').length).toBe(2);
});
it('renders multi-line \\[ \\] as display math', () => {
expect(render('Then:\n\n\\[\ns = ut\n\\]\n\nDone.').display).toBe(1);
expect(render('\\[\na+b=c\n\\]').display).toBe(1);
});
it('does not read math lines as markdown structure', () => {
// A lone `=` under a line is a setext heading and a `- ` line a list item,
// outside math.
const out = render('\\[\nx\n=\ny\n- z\n\\]');
expect(out.display).toBe(1);
expect(container.querySelector('h1, ul')).toBeNull();
expect(formulas()).toEqual(['x\n=\ny\n- z']);
});
it('renders mid-sentence \\[ \\] as math', () => {
const out = render('So \\[ s = ut \\] holds.');
expect(out.inline + out.display).toBe(1);
expect(out.text).toContain('holds.');
});
it('keeps the rest of the answer when a line opens with \\[ \\] and goes on', () => {
// A line-start `\[` is a display block only when its `\]` ends the line;
// otherwise it must not swallow everything after it.
const out = render('\\[a+b=c\\] and more text\n\nNext paragraph.');
expect(out.inline).toBe(1);
expect(out.display).toBe(0);
expect(out.text).toContain('and more text');
expect(container.querySelectorAll('p').length).toBe(2);
});
it.each([
['\\[x=1\\tag{1}\\]', '\\tag'],
['\\[\\begin{aligned}x&=1\\end{aligned}\\]', 'aligned'],
['where \\(x = 1 \\tag{2}\\) holds', 'inline \\tag'],
])('renders %s as display math', (source) => {
const out = render(source);
expect(out.display).toBe(1);
expect(out.errors).toBe(0);
});
it('renders display math inside a blockquote', () => {
const out = render('> \\[\n> x\n> =\n> y\n> \\]');
expect(out.display).toBe(1);
expect(container.querySelector('blockquote .katex-display')).not.toBeNull();
expect(formulas()).toEqual(['x\n=\ny']);
});
it('renders display math inside a list item', () => {
const out = render('- \\[\n x\n =\n y\n \\]');
expect(out.display).toBe(1);
expect(container.querySelector('li .katex-display')).not.toBeNull();
});
it('does not close a display block across the end of a blockquote', () => {
const out = render('> \\[\n> x\n> =\n\ny\n\\]\n\nNext section\n=');
expect(out.inline + out.display + out.errors).toBe(0);
});
it('does not render an unclosed \\( as math', () => {
const out = render('Using \\(v = u');
expect(out.inline + out.display + out.errors).toBe(0);
});
it('does not render an unclosed \\[ block as math', () => {
const out = render('Then\n\n\\[\nv = u\n\nand the rest.');
expect(out.inline + out.display + out.errors).toBe(0);
expect(out.text).toContain('and the rest.');
});
it('keeps an escaped task checkbox as text', () => {
const out = render('- \\[ \\] todo\n- done');
expect(out.inline + out.display + out.errors).toBe(0);
expect(out.text).toContain('todo');
expect(container.querySelectorAll('li').length).toBe(2);
});
it('keeps escaped index brackets as text', () => {
const out = render('arr\\[0\\] and arr\\[1\\]');
expect(out.inline + out.display).toBe(0);
expect(out.text).toContain('arr[0] and arr[1]');
});
it('keeps nested \\( \\) in one span', () => {
// KaTeX cannot typeset a nested `\(`, but the span must not end at the
// inner `\)` and leak the outer one into the text.
const out = render('\\(a + \\(b + c\\)\\) end');
expect(out.inline + out.errors).toBe(1);
expect(out.text.trim()).toBe('end');
});
it('does not mistake a \\[ \\] after inline code for its own line', () => {
const out = render('Use `f` \\[x\\]\nnext');
expect(out.inline + out.display).toBe(1);
expect(out.text).not.toContain('$');
});
it('leaves \\( \\) inside code untouched', () => {
const out = render('Type `\\(x\\)` literally.');
expect(out.inline).toBe(0);
expect(container.querySelector('code')?.textContent).toBe('\\(x\\)');
});
it('leaves bash prompt escapes in a fence untouched', () => {
const out = render('```bash\nPS1="\\[\\e[0m\\]$ "\n```');
expect(out.inline + out.display).toBe(0);
expect(container.textContent).toContain('PS1="\\[\\e[0m\\]$ "');
});
});
describe('dollar delimiters', () => {
it.each([
'from $2bn to at least $4bn per operation starting Sept 9.',
'Price is $50 and $100',
'$50 is $20 + $30',
'Total: $29.50 plus tax',
'Revenue: $5M to $10M, funding: $1.5B, price: $5K',
'$250k is 25% of $1M',
'Bitcoin: $0.00001234, Gas: $3.999, Rate: $1.234567890',
'The total is $1157.90 (existing) + $500 (new investment) = $1657.90.',
'a $100-$200 range',
'a $100–$200 range',
'in the $10k-$20k band',
'- **Total Savings**: $500 + $200 + $150 = $850',
'Cela coûte 100$ et 200$ en Europe',
'A single $ sign should not be converted',
'The price hit $79,455 on',
'Currency $100 and\nthen $200 later',
'The cost is $',
'Use $variable in the code',
'| Date | Price | Note |\n|---|---|---|\n| Aug 21 | $77,300, peak $79,455 | White House Clarity Act push |',
'It costs $5 and $10 in total.',
'Plans range between $5-$10 per seat.',
'Price: $19.99/month or $199/year.',
'Revenue grew from $3.2B to $4.1B.',
'Price is $5 (was $10).',
'| Plan | Price |\n|---|---|\n| Basic | $5 |\n| Pro | $10 |',
])('leaves currency as text: %s', (source) => {
const out = render(source);
expect(out.inline + out.display + out.errors).toBe(0);
const amount = source.match(/\$\d[\d.,]*|\d+\$/);
if (amount) expect(out.text).toContain(amount[0]);
});
it.each([
['Inline math: $x^2 + y^2 = z^2$', 1],
['the answer is $3$.', 1],
['an eigenvalue of $-1$ is expected', 1],
['the $n$th term', 1],
['The set is defined as $\\{x | x > 0\\}$.', 1],
['Calculate $\\text{Total} = \\$500 + \\$200$', 1],
['The equation $12 \\times 12 = 144$ is simple', 1],
['Price $100 then equation $x + y = z$ then another price $50', 1],
['Formula $x^2$ costs $25', 1],
])('renders %s as math', (source, count) => {
const out = render(source);
expect(out.inline).toBe(count);
expect(out.errors).toBe(0);
});
it('keeps the price beside a formula', () => {
render('Cost is $5 per unit and $x$ units.');
expect(formulas()).toEqual(['x']);
expect(container.textContent).toContain('$5 per unit');
});
it('does not form emphasis out of math', () => {
const out = render('terms $a_1 + b_2$ and $c_{i}^{*}$ here');
expect(out.inline).toBe(2);
expect(container.querySelector('em, strong')).toBeNull();
});
it('renders chemistry with mhchem', () => {
const out = render('$\\ce{H2O}$ and $\\pu{123 J}$');
expect(out.inline).toBe(2);
expect(out.errors).toBe(0);
});
it.each([
[
'a code span',
'The error "invalid $lookup namespace" occurs when using `$lookup` operator',
],
['escaped dollars', 'Already escaped \\$50 and \\$100'],
['a line break', 'This has $x\ny$ which spans lines'],
['an unbalanced brace', 'weird $a}b$ y'],
[
'shell PIDs in a fence',
'```bash\npstree -p $$\necho $$\n```\n\nSome text after.',
],
])('renders no math across %s', (_, source) => {
const out = render(source);
expect(out.inline + out.display + out.errors).toBe(0);
});
it('renders math only outside a fence', () => {
const out = render('```\n$100\n$variable\n```\n\nOutside $x^2$');
expect(out.inline).toBe(1);
expect(container.querySelector('code')?.textContent).toContain('$variable');
});
it('renders a $$ paragraph as display math', () => {
const out = render('text\n\n$$E=mc^2$$\n\nmore');
expect(out.display).toBe(1);
expect(out.text).toContain('more');
});
it('keeps mid-line $$ math inline', () => {
const out = render('so $$E=mc^2$$ holds');
expect(out.inline).toBe(1);
expect(out.display).toBe(0);
});
it('renders $$ with \\begin on the fence line', () => {
const out = render(
'$$\\begin{aligned}\na&=b\\\\\nc&=d\n\\end{aligned}$$\n\nAfter.',
);
expect(out.display).toBe(1);
expect(out.errors).toBe(0);
expect(out.text).toContain('After.');
});
});
describe('citations', () => {
it('links citations next to inline math and renders the math', () => {
const out = render('Newton: \\(F=ma\\) [1] and \\(p=mv\\)[2].');
expect(out.inline).toBe(2);
expect(citations()).toEqual(['1', '2']);
});
it('never links a bracket inside math', () => {
const out = render('$$a[1]$$ and [3]');
expect(formulas()).toEqual(['a[1]']);
expect(out.errors).toBe(0);
expect(citations()).toEqual(['3']);
});
it('never links a bracket inside single-dollar math', () => {
const out = render('value $a[1]$ end [2]');
expect(citations()).toEqual(['2']);
expect(out.errors).toBe(0);
});
it.each([
['**[1]** and [2][3]', ['1', '2', '3']],
['see `row[0]` and [4]', ['4']],
['[link](https://x.com) [1]', ['1']],
['- one\n - nested [1]', ['1']],
['Per [1], run:\n```js\nconst a = arr[0];\n```\nthen [2].', ['1', '2']],
])('links %s', (source, expected) => {
render(source);
expect(citations()).toEqual(expected);
});
it('leaves code and diagram labels alone', () => {
render('```python\nprint(row[0])\n```');
expect(container.textContent).toContain('print(row[0])');
expect(citations()).toEqual([]);
});
it('keeps a reference-style link a link', () => {
render('See [the docs][1].\n\n[1]: https://example.com');
expect(
container.querySelector('a[href="https://example.com"]'),
).not.toBeNull();
expect(citations()).toEqual([]);
});
it('links only citations within the number of sources', () => {
render('Per [1], [2] and [0] and [7].', { sourceCount: 2 });
expect(citations()).toEqual(['1', '2']);
expect(container.textContent).toContain('[0] and [7]');
});
it('links no citation when the answer has no sources', () => {
render('Per [1].', { sourceCount: 0 });
expect(citations()).toEqual([]);
expect(container.textContent).toContain('[1]');
});
});
describe('citation pills', () => {
function click(label: string) {
const pill = Array.from(container.querySelectorAll('button')).find(
(node) => node.textContent === label,
);
act(() => pill?.click());
}
it('scrolls to the source card when it is shown', () => {
const card = document.createElement('div');
card.id = 'source-1';
card.scrollIntoView = vi.fn();
document.body.appendChild(card);
const onOpenSources = vi.fn();
try {
render('Per [2].', { sourceCount: 5, onOpenSources });
click('2');
expect(card.scrollIntoView).toHaveBeenCalled();
expect(onOpenSources).not.toHaveBeenCalled();
} finally {
card.remove();
}
});
it('opens the sources list for a source with no card', () => {
// Only the first three sources get a card; `[5]` has none to scroll to.
const onOpenSources = vi.fn();
render('Per [5].', { sourceCount: 5, onOpenSources });
click('5');
expect(onOpenSources).toHaveBeenCalledTimes(1);
});
});
describe('mermaid', () => {
it('renders a closed mermaid fence as a diagram, source untouched', () => {
render(
'Intro [1]\n\n```mermaid\nflowchart TD\n A[0] --> B[1]\n```\n\nOutro',
);
expect(
container.querySelector('[data-testid="mermaid"]')?.textContent,
).toBe('flowchart TD\n A[0] --> B[1]');
expect(citations()).toEqual(['1']);
expect(container.textContent).toContain('Outro');
});
it('shows an unclosed mermaid fence as code while it streams', () => {
render('```mermaid\nflowchart TD\n A --> B', { isStreaming: true });
expect(container.querySelector('[data-testid="mermaid"]')).toBeNull();
});
});
describe('streaming', () => {
it('closes an open $$ block while streaming', () => {
const out = render('$$\nx = 1\ny = 2', { isStreaming: true });
expect(out.display).toBe(1);
});
it('closes an open inline $$ while streaming', () => {
const out = render('Text with $$formula', { isStreaming: true });
expect(out.inline).toBe(1);
expect(out.text).not.toContain('$');
});
it('closes an open \\[ block while streaming', () => {
const out = render('Then:\n\n\\[\nv = u', { isStreaming: true });
expect(out.display).toBe(1);
});
it('closes open bold before an open formula, not after it', () => {
// Appended to the end, `**` turned the closing fence into `$$**`.
const out = render('Energy is **the work it takes\n$$\nE = mc', {
isStreaming: true,
});
expect(out.display).toBe(1);
expect(out.errors).toBe(0);
expect(formulas()).toEqual(['E = mc']);
});
it('holds back an inline $$ with nothing after it yet', () => {
const out = render('Then **bold** and $$', { isStreaming: true });
expect(out.inline + out.display + out.errors).toBe(0);
expect(out.text).not.toContain('$');
});
it('does not render an unclosed \\( while streaming', () => {
const out = render('Using \\(v = u', { isStreaming: true });
expect(out.inline + out.display + out.errors).toBe(0);
});
it('leaves a trailing $ alone', () => {
const out = render('The cost is $', { isStreaming: true });
expect(out.inline + out.display).toBe(0);
expect(out.text).toContain('The cost is $');
});
it('closes open bold while streaming', () => {
render('Text **bold', { isStreaming: true });
expect(container.querySelector('strong')?.textContent).toBe('bold');
});
it('does not add $$ inside an open code fence', () => {
render('```bash\necho $$', { isStreaming: true });
expect(container.querySelector('code')?.textContent).toBe('echo $$');
});
it('does not heal once the answer is complete', () => {
render('Text **bold');
expect(container.querySelector('strong')).toBeNull();
});
it('shows a formula that cannot render yet in the muted colour', () => {
const out = render('$$\n\\frac{a}{', { isStreaming: true });
expect(out.errors).toBe(1);
const error = container.querySelector<HTMLElement>('.katex-error');
expect(error?.getAttribute('style')).toContain('var(--muted-foreground)');
});
});
describe('block rendering', () => {
it('re-renders only the block that changed', () => {
render('First \\(x^2\\) [1].\n\nSecond', { isStreaming: true });
markdownRenders.mockClear();
render('First \\(x^2\\) [1].\n\nSecond paragraph grows', {
isStreaming: true,
});
// The first paragraph, formula and all, is not parsed or typeset again.
expect(markdownRenders.mock.calls).toEqual([['Second paragraph grows']]);
expect(container.textContent).toContain('Second paragraph grows');
});
it('keeps footnotes working across blocks', () => {
render('Claim[^1].\n\nMore.\n\n[^1]: The note.');
expect(
container.querySelector('a[href="#user-content-fn-1"]'),
).not.toBeNull();
});
});
+294 -316
View File
@@ -1,16 +1,17 @@
import 'katex/dist/katex.min.css';
// Registers `\ce` and `\pu` with the KaTeX that rehype-katex renders with.
import 'katex/contrib/mhchem';
import { Fragment, type ReactNode, useMemo } from 'react';
import { Fragment, memo, type ReactNode, useMemo } from 'react';
import { useTranslation } from 'react-i18next';
import ReactMarkdown from 'react-markdown';
import ReactMarkdown, { type Components } from 'react-markdown';
import { Prism as SyntaxHighlighter } from 'react-syntax-highlighter';
import {
oneLight,
vscDarkPlus,
} from 'react-syntax-highlighter/dist/cjs/styles/prism';
import rehypeKatex from 'rehype-katex';
import remarkGfm from 'remark-gfm';
import remarkMath from 'remark-math';
import type { PluggableList } from 'unified';
import { markdownHeadings } from '@/lib/markdown';
@@ -19,6 +20,14 @@ import MermaidRenderer from '../components/MermaidRenderer';
import { Button } from '../components/ui/button';
import { useDarkTheme } from '../hooks';
import classes from './ConversationBubble.module.css';
import {
answerSyntaxPlugins,
healStreamingMarkdown,
normalizeMathFences,
splitAnswerBlocks,
} from './markdown/answerMarkdown';
import { remarkCitations } from './markdown/remarkCitations';
import { remarkDisplayMath } from './markdown/remarkDisplayMath';
import {
resolveSandboxLink,
type SandboxArtifact,
@@ -26,118 +35,63 @@ import {
} from './sandboxLinks';
import { cn } from '@/lib/utils';
// One fenced block or inline code span. Backtick runs are length-matched, so a
// ```` fence closes only on ```` and nested fences stay masked. The
// unterminated alternatives keep an open span matched too: answers stream in,
// so a fence is open for most of its life and needs protecting the whole time,
// not just once the closing fence arrives. A backtick fence's info string may
// not itself contain backticks, so a line that opens with an inline span stays
// an inline span. Four-space indented blocks are deliberately not masked —
// list continuation lines are indented the same way, and masking those would
// drop real citations out of nested lists.
const CODE_SPAN =
/(?<![^\n])[ \t]*(`{3,})[^`\n]*(?![^\n])(?:[\s\S]*?\n[ \t]*\1`*[ \t]*\r?(?![^\n])|[\s\S]*$)|(?<![^\n])[ \t]*(~{3,})[^\n]*(?:[\s\S]*?\n[ \t]*\2~*[ \t]*\r?(?![^\n])|[\s\S]*$)|(`+)(?:[^\n]|\n(?![ \t\r]*\n))*?(?<!`)\3(?!`)|`+[^`\n]*$/g;
// A formula KaTeX cannot typeset (a half-streamed one, most often) shows its
// source in the muted text colour rather than KaTeX's red.
const rehypePlugins: PluggableList = [
[rehypeKatex, { errorColor: 'var(--muted-foreground)' }],
];
// ``\[ \]`` and ``\( \)`` LaTeX, which remark-math does not recognise.
const LATEX_SPAN = /\\\[[\s\S]*?\\\]|\\\([\s\S]*?\\\)/g;
// Applies `transform` to everything outside the regions `mask` matches, and
// `onMasked` to those regions themselves.
function transformOutside(
content: string,
mask: RegExp,
transform: (segment: string) => string,
onMasked: (segment: string) => string = (segment) => segment,
): string {
const pattern = new RegExp(mask.source, mask.flags); // its own lastIndex
let result = '';
let index = 0;
let match: RegExpExecArray | null;
while ((match = pattern.exec(content)) !== null) {
result += transform(content.slice(index, match.index)) + onMasked(match[0]);
index = pattern.lastIndex;
}
return result + transform(content.slice(index));
}
// Runs `transform` over the prose in `content`, leaving code untouched.
//
// The rewrites below are plain regex over the whole document, and both used to
// corrupt the user's own code: `print(row[0])` rendered as
// `print(row[0](#cite-0))`, and `PS1="\[\e[0m\]"` as `PS1="$$\e[0m$$"`. Neither
// can move into a remark plugin: remark-math tokenises `$`/`$$` while parsing,
// so the LaTeX pass has to run on the raw string before remark ever sees it.
export function applyOutsideCode(
content: string,
transform: (segment: string) => string,
): string {
return transformOutside(content, CODE_SPAN, transform);
}
// Rewrites one LaTeX span into the ``$$``/``$`` form remark-math parses.
function toDollarMath(span: string): string {
const equation = span.slice(2, -2);
return span.startsWith('\\[') ? `$$${equation}$$` : `$${equation}$`;
}
// Replaces block-level ``\[ \]`` and inline ``\( \)`` LaTeX delimiters with the
// ``$$``/``$`` forms remark-math understands.
export function preprocessLaTeX(content: string): string {
return applyOutsideCode(content, (prose) =>
prose.replace(LATEX_SPAN, toDollarMath),
// One top-level block of the answer. Memoised, so while an answer streams only
// the block that is still growing is parsed and typeset again.
const MarkdownBlock = memo(function MarkdownBlock({
content,
remarkPlugins,
components,
}: {
content: string;
remarkPlugins: PluggableList;
components: Components;
}) {
return (
<ReactMarkdown
remarkPlugins={remarkPlugins}
rehypePlugins={rehypePlugins}
urlTransform={sandboxUrlTransform}
components={components}
>
{content}
</ReactMarkdown>
);
}
});
// Turns citation references ``[N]`` into ``[N](#cite-N)`` links so
// ReactMarkdown renders them as <a> tags we can style. The lookarounds skip
// references that are already links.
function linkCitations(prose: string): string {
return prose.replace(
/(?<!\[)\[(\d+)\](?!\()/g,
(_, num) => `[${num}](#cite-${num})`,
// Artifact lists are rebuilt on every render of the bubble; keep one identity
// per content so the memoised blocks are not re-rendered for nothing.
function useStableArtifacts(list?: SandboxArtifact[]) {
const signature = JSON.stringify(
list?.map(({ id, label, toolName, ref }) => [id, label, toolName, ref]),
);
return useMemo(() => list, [signature]);
}
type ContentSegment = { type: 'text' | 'mermaid'; content: string };
export function processMarkdownContent(content: string): ContentSegment[] {
// Citations are linked outside code — an index subscript like `row[0]` is not
// a citation — and outside the math this same pass produces. Math the answer
// already wrote as ``$…$`` stays unmasked, since ``$`` is also a currency
// sign. This has to run before the ```mermaid split below, so diagram source
// needs the same guard.
const processedContent = applyOutsideCode(content, (prose) =>
transformOutside(prose, LATEX_SPAN, linkCitations, toDollarMath),
);
const contentSegments: ContentSegment[] = [];
let lastIndex = 0;
const regex = /```mermaid\n([\s\S]*?)```/g;
let match;
while ((match = regex.exec(processedContent)) !== null) {
const textBefore = processedContent.substring(lastIndex, match.index);
if (textBefore) contentSegments.push({ type: 'text', content: textBefore });
contentSegments.push({ type: 'mermaid', content: match[1].trim() });
lastIndex = match.index + match[0].length;
}
const textAfter = processedContent.substring(lastIndex);
if (textAfter) contentSegments.push({ type: 'text', content: textAfter });
return contentSegments;
}
type AnswerGroup =
{ type: 'markdown'; blocks: string[] } | { type: 'mermaid'; content: string };
export default function MarkdownAnswer({
content,
isStreaming,
artifacts,
turnArtifacts,
sourceCount,
artifacts: artifactsProp,
turnArtifacts: turnArtifactsProp,
onOpenArtifact,
onOpenSources,
}: {
content: string;
isStreaming?: boolean;
/**
* How many sources the answer cites from; ``[N]`` beyond it is left as
* text, with no source to jump to. Unset links every ``[N]``.
*/
sourceCount?: number;
/**
* Every artifact in the conversation, in creation order (``A1`` is the
* first) — refs are conversation-scoped, so a link may name an earlier turn's
@@ -147,233 +101,257 @@ export default function MarkdownAnswer({
/** This turn's own artifacts; the filename fallback prefers them. */
turnArtifacts?: SandboxArtifact[];
onOpenArtifact?: (artifact: { id: string; toolName: string }) => void;
/** Opens the full sources list, for a citation whose card is not shown. */
onOpenSources?: () => void;
}) {
const { t } = useTranslation();
const [isDarkTheme] = useDarkTheme();
// Re-runs on every streamed token otherwise.
const contentSegments = useMemo(
() => processMarkdownContent(content),
[content],
const artifacts = useStableArtifacts(artifactsProp);
const turnArtifacts = useStableArtifacts(turnArtifactsProp);
const groups = useMemo(() => {
const normalized = normalizeMathFences(content);
const blocks = splitAnswerBlocks(
isStreaming ? healStreamingMarkdown(normalized) : normalized,
);
// Consecutive markdown blocks share one column; a diagram breaks it.
const grouped: AnswerGroup[] = [];
for (const block of blocks) {
const last = grouped[grouped.length - 1];
if (block.type === 'mermaid') grouped.push(block);
else if (last?.type === 'markdown') last.blocks.push(block.content);
else grouped.push({ type: 'markdown', blocks: [block.content] });
}
return grouped;
}, [content, isStreaming]);
const remarkPlugins = useMemo<PluggableList>(
() => [
...answerSyntaxPlugins,
remarkDisplayMath,
[remarkCitations, { sourceCount }],
],
[sourceCount],
);
// Shared by the `a` and `img` renderers: a generated file is already on the
// turn as a download chip, so both point at the chip rather than at a URL no
// browser can open.
const renderArtifactChip = (
artifact: SandboxArtifact,
content: ReactNode,
) => {
if (!onOpenArtifact) return <>{content}</>;
return (
<Button
type="button"
variant="link"
size="inline"
onClick={() =>
onOpenArtifact({
id: artifact.id,
toolName: artifact.toolName ?? '',
})
const components = useMemo<Components>(() => {
// Shared by the `a` and `img` renderers: a generated file is already on the
// turn as a download chip, so both point at the chip rather than at a URL no
// browser can open.
const renderArtifactChip = (
artifact: SandboxArtifact,
content: ReactNode,
) => {
if (!onOpenArtifact) return <>{content}</>;
return (
<Button
type="button"
variant="link"
size="inline"
onClick={() =>
onOpenArtifact({
id: artifact.id,
toolName: artifact.toolName ?? '',
})
}
/* Sits mid-sentence, so it must wrap with the surrounding text. */
className="whitespace-normal"
>
{content}
</Button>
);
};
return {
...markdownHeadings,
a({ href, children }) {
// A generated file is already on the turn as a download
// chip, but the model links it with a `sandbox:`/`artifact:`
// URL no browser can open. Point the link at the chip
// instead, and never leave a dead anchor behind.
const sandboxLink = resolveSandboxLink(href, artifacts, turnArtifacts);
if (sandboxLink.kind === 'plain') {
return <>{children}</>;
}
/* Sits mid-sentence, so it must wrap with the surrounding text. */
className="whitespace-normal"
>
{content}
</Button>
);
};
if (sandboxLink.kind === 'artifact') {
return renderArtifactChip(sandboxLink.artifact, children);
}
if (href?.startsWith('#cite-')) {
const num = href.replace('#cite-', '');
const sourceIdx = parseInt(num, 10) - 1;
return (
<Button
type="button"
variant="secondary"
size="xs"
shape="pill"
onClick={() => {
const el = document.getElementById(`source-${sourceIdx}`);
if (el) {
el.scrollIntoView({
behavior: 'smooth',
block: 'center',
});
el.classList.add('ring-3', 'ring-primary');
setTimeout(
() => el.classList.remove('ring-3', 'ring-primary'),
2000,
);
} else {
// Only the first few sources get a card to scroll to.
onOpenSources?.();
}
}}
className="mx-0.5 h-5 min-w-5"
title={t('conversation.jumpToSource', { num })}
>
{num}
</Button>
);
}
return (
<a href={href} target="_blank" rel="noopener noreferrer">
{children}
</a>
);
},
img({ src, alt }) {
// `![chart](sandbox:/mnt/data/chart.png)` is how a model
// announces a plot it produced. react-markdown runs
// urlTransform over `src` too, so without this the invented
// scheme survives and renders a broken-image box beside the
// chip that opens the very same file.
const source = typeof src === 'string' ? src : undefined;
const sandboxLink = resolveSandboxLink(
source,
artifacts,
turnArtifacts,
);
if (sandboxLink.kind === 'artifact') {
const { artifact } = sandboxLink;
return renderArtifactChip(
artifact,
alt || artifact.label || t('conversation.openFile'),
);
}
if (sandboxLink.kind === 'plain') {
return <>{alt ?? ''}</>;
}
return <img src={source} alt={alt} className="max-w-full" />;
},
code(props) {
const { children, className, node, ref, ...rest } = props;
const match = /language-(\w+)/.exec(className || '');
const language = match ? match[1] : '';
return match ? (
<div className="group border-border relative overflow-hidden rounded-xl border">
<div className="bg-muted flex items-center justify-between px-2 py-1">
<span className="text-foreground text-xs font-medium">
{language}
</span>
<CopyButton
textToCopy={String(children).replace(/\n$/, '')}
side="bottom"
/>
</div>
<SyntaxHighlighter
{...rest}
PreTag="div"
language={language}
/* eslint-disable-next-line shadcn/no-inline-styles -- SyntaxHighlighter's style prop is its Prism theme object (oneLight / vscDarkPlus), picked by theme at runtime; it is not CSS. See DESIGN.md, Approved exceptions. */
style={isDarkTheme ? vscDarkPlus : oneLight}
className="mt-0!"
customStyle={{ margin: 0, borderRadius: 0 }}
>
{String(children).replace(/\n$/, '')}
</SyntaxHighlighter>
</div>
) : (
<code className="bg-accent text-foreground rounded-md px-2 py-1 text-xs font-normal whitespace-pre-line">
{children}
</code>
);
},
ul({ children }) {
return (
<ul
className={cn(
'list-inside list-disc pl-4 whitespace-normal',
classes.list,
)}
>
{children}
</ul>
);
},
ol({ children }) {
return (
<ol
className={cn(
'list-inside list-decimal pl-4 whitespace-normal',
classes.list,
)}
>
{children}
</ol>
);
},
table({ children }) {
return (
<div className="border-border relative overflow-x-auto rounded-lg border">
<table className="text-foreground w-full text-left">
{children}
</table>
</div>
);
},
thead({ children }) {
return (
<thead className="bg-muted text-foreground text-xs uppercase">
{children}
</thead>
);
},
tr({ children }) {
return (
<tr className="border-border odd:bg-card even:bg-muted border-b">
{children}
</tr>
);
},
th({ children }) {
return <th className="px-6 py-3">{children}</th>;
},
td({ children }) {
return <td className="px-6 py-3">{children}</td>;
},
};
}, [t, isDarkTheme, artifacts, turnArtifacts, onOpenArtifact, onOpenSources]);
return (
<>
{contentSegments.map((segment, index) => (
{groups.map((group, index) => (
<Fragment key={index}>
{segment.type === 'text' ? (
<div className="animate-in fade-in flex flex-col gap-3 leading-normal wrap-break-word whitespace-pre-wrap duration-160 ease-out motion-reduce:animate-none">
<ReactMarkdown
remarkPlugins={[
remarkGfm,
[remarkMath, { singleDollarTextMath: false }],
]}
rehypePlugins={[rehypeKatex]}
urlTransform={sandboxUrlTransform}
components={{
...markdownHeadings,
a({ href, children }) {
// A generated file is already on the turn as a download
// chip, but the model links it with a `sandbox:`/`artifact:`
// URL no browser can open. Point the link at the chip
// instead, and never leave a dead anchor behind.
const sandboxLink = resolveSandboxLink(
href,
artifacts,
turnArtifacts,
);
if (sandboxLink.kind === 'plain') {
return <>{children}</>;
}
if (sandboxLink.kind === 'artifact') {
return renderArtifactChip(sandboxLink.artifact, children);
}
if (href?.startsWith('#cite-')) {
const num = href.replace('#cite-', '');
const sourceIdx = parseInt(num, 10) - 1;
return (
<Button
type="button"
variant="secondary"
size="xs"
shape="pill"
onClick={() => {
const el = document.getElementById(
`source-${sourceIdx}`,
);
if (el) {
el.scrollIntoView({
behavior: 'smooth',
block: 'center',
});
el.classList.add('ring-3', 'ring-primary');
setTimeout(
() =>
el.classList.remove('ring-3', 'ring-primary'),
2000,
);
}
}}
className="mx-0.5 h-5 min-w-5"
title={t('conversation.jumpToSource', { num })}
>
{num}
</Button>
);
}
return (
<a href={href} target="_blank" rel="noopener noreferrer">
{children}
</a>
);
},
img({ src, alt }) {
// `![chart](sandbox:/mnt/data/chart.png)` is how a model
// announces a plot it produced. react-markdown runs
// urlTransform over `src` too, so without this the invented
// scheme survives and renders a broken-image box beside the
// chip that opens the very same file.
const source = typeof src === 'string' ? src : undefined;
const sandboxLink = resolveSandboxLink(
source,
artifacts,
turnArtifacts,
);
if (sandboxLink.kind === 'artifact') {
const { artifact } = sandboxLink;
return renderArtifactChip(
artifact,
alt || artifact.label || t('conversation.openFile'),
);
}
if (sandboxLink.kind === 'plain') {
return <>{alt ?? ''}</>;
}
return (
<img src={source} alt={alt} className="max-w-full" />
);
},
code(props) {
const { children, className, node, ref, ...rest } = props;
const match = /language-(\w+)/.exec(className || '');
const language = match ? match[1] : '';
return match ? (
<div className="group border-border relative overflow-hidden rounded-xl border">
<div className="bg-muted flex items-center justify-between px-2 py-1">
<span className="text-foreground text-xs font-medium">
{language}
</span>
<CopyButton
textToCopy={String(children).replace(/\n$/, '')}
side="bottom"
/>
</div>
<SyntaxHighlighter
{...rest}
PreTag="div"
language={language}
/* eslint-disable-next-line shadcn/no-inline-styles -- SyntaxHighlighter's style prop is its Prism theme object (oneLight / vscDarkPlus), picked by theme at runtime; it is not CSS. See DESIGN.md, Approved exceptions. */
style={isDarkTheme ? vscDarkPlus : oneLight}
className="mt-0!"
customStyle={{ margin: 0, borderRadius: 0 }}
>
{String(children).replace(/\n$/, '')}
</SyntaxHighlighter>
</div>
) : (
<code className="bg-accent text-foreground rounded-md px-2 py-1 text-xs font-normal whitespace-pre-line">
{children}
</code>
);
},
ul({ children }) {
return (
<ul
className={cn(
'list-inside list-disc pl-4 whitespace-normal',
classes.list,
)}
>
{children}
</ul>
);
},
ol({ children }) {
return (
<ol
className={cn(
'list-inside list-decimal pl-4 whitespace-normal',
classes.list,
)}
>
{children}
</ol>
);
},
table({ children }) {
return (
<div className="border-border relative overflow-x-auto rounded-lg border">
<table className="text-foreground w-full text-left">
{children}
</table>
</div>
);
},
thead({ children }) {
return (
<thead className="bg-muted text-foreground text-xs uppercase">
{children}
</thead>
);
},
tr({ children }) {
return (
<tr className="border-border odd:bg-card even:bg-muted border-b">
{children}
</tr>
);
},
th({ children }) {
return <th className="px-6 py-3">{children}</th>;
},
td({ children }) {
return <td className="px-6 py-3">{children}</td>;
},
}}
>
{segment.content}
</ReactMarkdown>
{group.type === 'markdown' ? (
<div
className={cn(
'animate-in fade-in flex flex-col gap-3 leading-normal wrap-break-word whitespace-pre-wrap duration-160 ease-out motion-reduce:animate-none',
classes.answer,
)}
>
{group.blocks.map((block, blockIndex) => (
<MarkdownBlock
key={blockIndex}
content={block}
remarkPlugins={remarkPlugins}
components={components}
/>
))}
</div>
) : (
<div className="my-4 w-full min-w-full">
<MermaidRenderer code={segment.content} isLoading={isStreaming} />
<MermaidRenderer code={group.content} isLoading={isStreaming} />
</div>
)}
</Fragment>
@@ -0,0 +1,241 @@
/**
* The string-level steps of rendering an answer, around the markdown parser.
*
* Math, code and citations are told apart by the parser (see `texMath.ts`,
* `singleDollarMath.ts` and the remark plugins beside them). What is left
* here works on the raw string only where the parser cannot help:
*
* - `normalizeMathFences` moves formula text off a ``$$`` fence line, which
* remark-math would read as the fence's metadata;
* - `healStreamingMarkdown` closes what a half-streamed answer has left open;
* - `splitAnswerBlocks` cuts the answer into its top-level blocks, so a
* streaming answer re-renders only the block that is still growing.
*/
import type { Root, RootContent } from 'mdast';
import remarkGfm from 'remark-gfm';
import remarkMath from 'remark-math';
import remarkParse from 'remark-parse';
import remend from 'remend';
import { type PluggableList, unified } from 'unified';
import { remarkSingleDollarMath } from './singleDollarMath';
import { remarkTexMath } from './texMath';
/** Syntax plugins; every parse of an answer, split or render, uses them. */
export const answerSyntaxPlugins: PluggableList = [
remarkGfm,
[remarkMath, { singleDollarTextMath: false }],
remarkTexMath,
remarkSingleDollarMath,
];
// A line's container prefix (blockquote markers, indentation) and the rest.
const LINE = /^((?:[ \t]*>)*[ \t]*)([\s\S]*)$/;
const FENCE_OPEN = /^(`{3,}|~{3,})/;
const DOLLAR_FENCE = /^\$\$\s*$/;
function splitLine(line: string): [prefix: string, rest: string] {
const [, prefix, rest] = LINE.exec(line) ?? ['', '', line];
return [prefix, rest];
}
function closesFence(rest: string, fence: string): boolean {
const run = rest.trim();
return (
run.length >= fence.length && run[0] === fence[0] && /^(?:`+|~+)$/.test(run)
);
}
/**
* Puts a ``$$`` fence's formula on its own line: ``$$\begin{aligned}`` opens a
* block whose first line remark-math reads as metadata and drops, and a
* closing ``\end{aligned}$$`` never closes it, so the block ran to the end of
* the answer. Code fences are left alone.
*/
export function normalizeMathFences(content: string): string {
if (!content.includes('$$')) return content;
const out: string[] = [];
let fence: string | null = null;
let inMath = false;
for (const line of content.split('\n')) {
const [prefix, rest] = splitLine(line);
if (fence) {
if (closesFence(rest, fence)) fence = null;
out.push(line);
continue;
}
if (inMath) {
const closing = /^(.*\S)[ \t]*\$\$\s*$/.exec(rest);
if (DOLLAR_FENCE.test(rest)) {
inMath = false;
} else if (closing) {
out.push(prefix + closing[1], `${prefix}$$`);
inMath = false;
continue;
}
out.push(line);
continue;
}
const codeFence = FENCE_OPEN.exec(rest);
const opening = /^\$\$(?!\$)(.*\S)\s*$/.exec(rest);
if (codeFence) {
fence = codeFence[1];
} else if (DOLLAR_FENCE.test(rest)) {
inMath = true;
} else if (opening && !opening[1].includes('$$')) {
out.push(`${prefix}$$`, prefix + opening[1].trimStart());
inMath = true;
continue;
}
out.push(line);
}
return out.join('\n');
}
// Inline code in a stretch of prose, open spans included, so a `$$` in it is
// not counted as math.
const INLINE_CODE = /(`+)[\s\S]*?(?:\1|$)/g;
// Splits a half-streamed answer at the math it is in the middle of writing:
// `head` is everything before that formula, `math` the formula closed. A ``$$``
// or line-start ``\[`` block gets its closing fence, an inline ``$$`` its
// closing ``$$``, and one with nothing after it yet is held back. Inside a code
// fence nothing is touched: a shell's `$$` is not math.
function splitOpenMath(content: string): { head: string; math: string } {
let fence: string | null = null;
let block: { close: string; prefix: string; start: number } | null = null;
let offset = 0;
for (const line of content.split('\n')) {
const [prefix, rest] = splitLine(line);
if (fence) {
if (closesFence(rest, fence)) fence = null;
} else if (block) {
const closed =
block.close === '$$' ? DOLLAR_FENCE.test(rest) : /\\\]\s*$/.test(rest);
if (closed) block = null;
} else if (FENCE_OPEN.test(rest)) {
fence = FENCE_OPEN.exec(rest)![1];
} else if (DOLLAR_FENCE.test(rest)) {
block = { close: '$$', prefix, start: offset };
} else if (rest.startsWith('\\[') && !rest.includes('\\]')) {
block = { close: '\\]', prefix, start: offset };
}
offset += line.length + 1;
}
if (fence) return { head: content, math: '' };
if (block) {
return {
head: content.slice(0, block.start),
math: `${content.slice(block.start)}\n${block.prefix}${block.close}`,
};
}
const paragraphStart = content.lastIndexOf('\n\n') + 1;
const prose = content
.slice(paragraphStart)
.replace(INLINE_CODE, '')
.replace(/\\\$/g, '');
const dollars = prose.match(/\$\$/g)?.length ?? 0;
if (dollars % 2 === 0) return { head: content, math: '' };
const opener = content.lastIndexOf('$$');
const formula = content.slice(opener);
return {
head: content.slice(0, opener),
math: formula.slice(2).trim() ? `${formula}$$` : '',
};
}
const REMEND_OPTIONS = {
// Math is closed above, code-aware; remend's own pass appends `$$` inside
// an open code fence.
katex: false,
inlineKatex: false,
// A half-typed `[1` or `\[` is not a link, and remend would strip its `[`.
links: false,
images: false,
// These two rewrite text that is already complete, so the answer would
// change once it stops streaming.
comparisonOperators: false,
singleTilde: false,
};
/**
* Closes what a half-streamed answer leaves open (``**bold``, `` `code``,
* ``$$`` and ``\[`` math), so it renders as it will once complete instead of
* flashing raw markdown. Only ever applied while the answer streams.
*/
export function healStreamingMarkdown(content: string): string {
// Emphasis is closed before the open formula, never after it: appended to
// the closing fence (``$$**``), it would stop the fence from closing.
const { head, math } = splitOpenMath(content);
if (!math) return remend(head, REMEND_OPTIONS);
const lineBreak = head.match(/\n*$/)![0];
return (
remend(head.slice(0, head.length - lineBreak.length), REMEND_OPTIONS) +
lineBreak +
math
);
}
export type AnswerBlock =
{ type: 'markdown'; content: string } | { type: 'mermaid'; content: string };
const parser = unified().use(remarkParse).use(answerSyntaxPlugins);
// Where a block's first line starts, so an indented block keeps its indent.
function lineStart(content: string, offset: number): number {
const start = content.lastIndexOf('\n', offset - 1) + 1;
return /^[ \t]*$/.test(content.slice(start, offset)) ? start : offset;
}
// A mermaid fence is drawn once it is closed; until then it streams as code.
function isDiagram(node: RootContent, source: string): boolean {
if (node.type !== 'code' || node.lang !== 'mermaid') return false;
const lines = source.trimEnd().split('\n');
const opening = FENCE_OPEN.exec(lines[0].trim());
return (
lines.length > 1 &&
opening !== null &&
closesFence(lines[lines.length - 1], opening[1])
);
}
/**
* Cuts an answer into its top-level blocks, each rendered on its own, and its
* closed mermaid fences, drawn as diagrams. An answer with link or footnote
* definitions stays whole between diagrams, since a reference resolves only
* against definitions in the same parse.
*/
export function splitAnswerBlocks(content: string): AnswerBlock[] {
const tree: Root = parser.parse(content);
const keepWhole = tree.children.some(
(node) => node.type === 'definition' || node.type === 'footnoteDefinition',
);
const blocks: AnswerBlock[] = [];
let groupStart: number | null = null;
for (const node of tree.children) {
const start = lineStart(content, node.position?.start.offset ?? 0);
const end = node.position?.end.offset ?? content.length;
if (node.type === 'code' && isDiagram(node, content.slice(start, end))) {
if (groupStart !== null) {
blocks.push({
type: 'markdown',
content: content.slice(groupStart, start),
});
groupStart = null;
}
blocks.push({ type: 'mermaid', content: node.value.trim() });
} else if (keepWhole) {
groupStart ??= start;
} else {
blocks.push({ type: 'markdown', content: content.slice(start, end) });
}
}
if (groupStart !== null) {
blocks.push({ type: 'markdown', content: content.slice(groupStart) });
}
return blocks;
}
@@ -0,0 +1,25 @@
import type { Extension } from 'micromark-util-types';
import type { Processor } from 'unified';
// The tokens `remark-math`'s `mathFromMarkdown` turns into math nodes, which
// the constructs here emit too.
declare module 'micromark-util-types' {
interface TokenTypeMap {
mathFlow: 'mathFlow';
mathFlowFence: 'mathFlowFence';
mathFlowFenceSequence: 'mathFlowFenceSequence';
mathFlowValue: 'mathFlowValue';
mathText: 'mathText';
mathTextData: 'mathTextData';
mathTextSequence: 'mathTextSequence';
}
}
/** Registers a micromark syntax extension from inside a remark plugin. */
export function addMicromarkExtension(
processor: Processor,
extension: Extension,
): void {
const data = processor.data() as { micromarkExtensions?: Extension[] };
(data.micromarkExtensions ??= []).push(extension);
}
@@ -0,0 +1,100 @@
/**
* Links source citations, ``[N]``, in the prose of an answer.
*
* Runs on the syntax tree, so only plain text is ever linked: code, inline
* code and math are their own nodes, never `text`, and link text is skipped.
* An index like `row[0]` in code, or `a[1]` in a formula, stays what it is.
* A bare ``[1]`` with no matching definition is text to the parser; a
* reference link ``[docs][1]`` with a ``[1]: url`` definition stays a link,
* and brackets the answer escaped, ``arr\[1\]``, stay brackets.
*
* The link targets ``#cite-N``, which `MarkdownAnswer` renders as the "jump to
* source" pill.
*/
import type { PhrasingContent, Root, Text } from 'mdast';
import { asciiPunctuation } from 'micromark-util-character';
import { SKIP, visit } from 'unist-util-visit';
export type CitationOptions = {
/**
* How many sources the answer has; only ``[1]`` to ``[sourceCount]`` are
* linked, since any other number has no source to jump to. Unset links
* every ``[N]``.
*/
sourceCount?: number;
};
const CITATION = /\[(\d+)\]/g;
// The offsets in `node.value` of characters the source wrote as a backslash
// escape. The parser resolves escapes, so ``\[1\]`` and ``[1]`` are the same
// text; only the source tells them apart. Anything else that makes the value
// differ from the source (an entity, a quote marker) ends the walk, and the
// rest of the node counts as unescaped.
function escapedOffsets(node: Text, source: string): Set<number> {
const escaped = new Set<number>();
const start = node.position?.start.offset;
const end = node.position?.end.offset;
if (start === undefined || end === undefined) return escaped;
const raw = source.slice(start, end);
if (!raw.includes('\\')) return escaped;
const { value } = node;
let i = 0;
for (let j = 0; j < value.length && i < raw.length; j++) {
// Indentation the parser dropped from a continuation line.
while (raw[i] !== value[j] && (raw[i] === ' ' || raw[i] === '\t')) i++;
if (
raw[i] === '\\' &&
raw[i + 1] === value[j] &&
asciiPunctuation(value.charCodeAt(j))
) {
escaped.add(j);
i += 2;
} else if (raw[i] === value[j]) {
i++;
} else {
break;
}
}
return escaped;
}
export function remarkCitations({ sourceCount }: CitationOptions = {}) {
const citable = (n: number) =>
sourceCount === undefined || (n >= 1 && n <= sourceCount);
return (tree: Root, file: { value?: unknown }) => {
const source = String(file.value ?? '');
visit(tree, (node, index, parent) => {
if (node.type === 'link' || node.type === 'linkReference') return SKIP;
if (node.type !== 'text' || !parent || index === undefined) return;
const escaped = escapedOffsets(node, source);
const parts: PhrasingContent[] = [];
let last = 0;
for (const match of node.value.matchAll(CITATION)) {
if (!citable(Number(match[1])) || escaped.has(match.index)) continue;
if (match.index > last) {
parts.push({
type: 'text',
value: node.value.slice(last, match.index),
});
}
parts.push({
type: 'link',
url: `#cite-${match[1]}`,
children: [{ type: 'text', value: match[1] }],
});
last = match.index + match[0].length;
}
if (parts.length === 0) return;
if (last < node.value.length) {
parts.push({ type: 'text', value: node.value.slice(last) });
}
parent.children.splice(index, 1, ...parts);
return [SKIP, index + parts.length];
});
};
}
@@ -0,0 +1,134 @@
/**
* Decides display versus inline math on the syntax tree, where the parser has
* already told math, code and prose apart.
*
* An inline formula becomes display math when
* - it is written with a block delimiter (``$$`` or ``\[``) and sits on a line
* of its own: ``$$E=mc^2$$`` alone in a paragraph, or a ``\[ … \]`` line the
* flow construct did not take (a lazy line inside a quote, say); or
* - KaTeX can only typeset it in display mode (``\tag``, or an environment such
* as ``align``); inline, it would be a red error instead of an equation.
*
* A promoted formula splits its paragraph: the text before and after stays a
* paragraph of its own. Where no paragraph can be split (a table cell, a
* heading) a display-only formula is still typeset in display mode, in place.
*/
import type { Paragraph, PhrasingContent, Root, RootContent } from 'mdast';
import { SKIP, visit } from 'unist-util-visit';
type InlineMath = Extract<PhrasingContent, { type: 'inlineMath' }>;
// `\tag` and the environments KaTeX refuses outside display mode.
const DISPLAY_ONLY =
/\\tag\b|\\begin\{(?:equation|align|alignat|gather|split|CD)\*?\}/;
const displayClasses = ['language-math', 'math-display'];
function isInlineMath(node: PhrasingContent): node is InlineMath {
return node.type === 'inlineMath';
}
function blockDelimited(node: InlineMath, source: string): boolean {
const start = node.position?.start.offset;
if (start === undefined) return false;
const opener = source.slice(start, start + 2);
return opener === '$$' || opener === '\\[';
}
// Whether the siblings either side of `index` end and start a line.
function aloneOnLine(children: PhrasingContent[], index: number): boolean {
const before = children[index - 1];
const after = children[index + 1];
const endsLine =
!before ||
before.type === 'break' ||
(before.type === 'text' && /(?:^|\n)[ \t]*$/.test(before.value));
const startsLine =
!after ||
after.type === 'break' ||
(after.type === 'text' && /^[ \t]*(?:\n|$)/.test(after.value));
return endsLine && startsLine;
}
function toDisplay(node: InlineMath): RootContent {
return {
type: 'math',
meta: null,
value: node.value,
position: node.position,
data: {
hName: 'pre',
hChildren: [
{
type: 'element',
tagName: 'code',
properties: { className: displayClasses },
children: [{ type: 'text', value: node.value }],
},
],
},
};
}
// The paragraph around each promoted formula, cut at it, with the line breaks
// that separated them dropped.
function splitParagraph(
paragraph: Paragraph,
promote: (child: PhrasingContent, index: number) => boolean,
): RootContent[] {
const blocks: RootContent[] = [];
let run: PhrasingContent[] = [];
let trimNext = false;
const flush = () => {
const last = run[run.length - 1];
if (last?.type === 'break') run.pop();
else if (last?.type === 'text') last.value = last.value.trimEnd();
run = run.filter((child) => child.type !== 'text' || child.value !== '');
if (run.length > 0) blocks.push({ type: 'paragraph', children: run });
run = [];
};
paragraph.children.forEach((child, index) => {
if (isInlineMath(child) && promote(child, index)) {
flush();
blocks.push(toDisplay(child));
trimNext = true;
return;
}
if (trimNext) {
trimNext = false;
if (child.type === 'break') return;
if (child.type === 'text') child.value = child.value.trimStart();
}
run.push(child);
});
flush();
return blocks;
}
export function remarkDisplayMath() {
return (tree: Root, file: { value?: unknown }) => {
const source = String(file.value);
visit(tree, 'paragraph', (node, index, parent) => {
if (!parent || index === undefined) return;
const promote = (child: PhrasingContent, at: number) =>
isInlineMath(child) &&
(DISPLAY_ONLY.test(child.value) ||
(blockDelimited(child, source) && aloneOnLine(node.children, at)));
if (!node.children.some(promote)) return;
const blocks = splitParagraph(node, promote);
// A paragraph's parent (the root, a list item, a quote) holds blocks.
(parent.children as RootContent[]).splice(index, 1, ...blocks);
return [SKIP, index + blocks.length];
});
// Nowhere to split (a table cell, a heading, inside bold): typeset a
// display-only formula in display mode where it stands.
visit(tree, 'inlineMath', (node) => {
if (!DISPLAY_ONLY.test(node.value)) return;
node.data = { ...node.data, hProperties: { className: displayClasses } };
});
};
}
@@ -0,0 +1,160 @@
/**
* Single-dollar inline math, ``$x$``, that leaves prices alone.
*
* Ported from LibreChat's `client/src/utils/latex.ts`
* (https://github.com/danny-avila/LibreChat), MIT License,
* Copyright (c) 2026 LibreChat.
*
* `$...$` is ambiguous with prices ("from $2bn to at least $4bn"), and only
* the tokenizer can decide a span without rewriting the message: code spans,
* fences and autolinks are structurally excluded, and a rejected span stays
* byte-identical text. `remark-math` keeps running with
* `singleDollarTextMath: false`; this construct is the only single-dollar path.
*
* A `$...$` span becomes math only when all of Pandoc's boundary rules hold:
* - the opening `$` is immediately followed by a non-space character;
* - the closing `$` is immediately preceded by a non-space character and not
* immediately followed by a digit (rejects ranges like "$100-$200");
* - the span stays on one line, contains no backtick, treats `\`-pairs as
* opaque (`\$` stays inside the span), and closes with balanced braces.
*
* A failed close abandons the whole attempt instead of scanning further, so
* the next `$` in "Price is $50 and $100" can never silently extend a span;
* micromark then retries the construct at that `$` on its own merits. Pandoc
* has the same known false positive: `$HOME/$USER` in prose (not code) is math.
*/
import {
asciiDigit,
markdownLineEnding,
markdownSpace,
} from 'micromark-util-character';
import { codes, types } from 'micromark-util-symbol';
import type {
Code,
Construct,
Effects,
Extension,
State,
TokenizeContext,
} from 'micromark-util-types';
import type { Processor } from 'unified';
import { addMicromarkExtension } from './micromarkExtension';
function tokenizeMathSpan(
this: TokenizeContext,
effects: Effects,
ok: State,
nok: State,
): State {
let previousCode: Code = null;
let braceDepth = 0;
return start;
function start(code: Code): State | undefined {
effects.enter('mathText');
effects.enter('mathTextSequence');
effects.consume(code);
effects.exit('mathTextSequence');
return open;
}
function open(code: Code): State | undefined {
if (
code === codes.eof ||
code === codes.dollarSign ||
code === codes.graveAccent ||
markdownSpace(code) ||
markdownLineEnding(code)
) {
return nok(code);
}
effects.enter('mathTextData');
return content(code);
}
function content(code: Code): State | undefined {
if (
code === codes.eof ||
code === codes.graveAccent ||
markdownLineEnding(code)
) {
return nok(code);
}
if (code === codes.dollarSign) {
if (markdownSpace(previousCode) || braceDepth !== 0) {
return nok(code);
}
effects.exit('mathTextData');
effects.enter('mathTextSequence');
effects.consume(code);
return close;
}
if (code === codes.backslash) {
effects.consume(code);
return escape;
}
if (code === codes.leftCurlyBrace) {
braceDepth++;
}
if (code === codes.rightCurlyBrace) {
if (braceDepth === 0) {
return nok(code);
}
braceDepth--;
}
previousCode = code;
effects.consume(code);
return content;
}
function escape(code: Code): State | undefined {
if (code === codes.eof || markdownLineEnding(code)) {
return nok(code);
}
previousCode = code;
effects.consume(code);
return content;
}
function close(code: Code): State | undefined {
if (code === codes.dollarSign || asciiDigit(code)) {
return nok(code);
}
effects.exit('mathTextSequence');
effects.exit('mathText');
return ok(code);
}
}
/** Mirrors `micromark-extension-math`: a `$` opener is valid unless it follows an unescaped `$`. */
function previous(this: TokenizeContext, code: Code): boolean {
return (
code !== codes.dollarSign ||
this.events[this.events.length - 1][1].type === types.characterEscape
);
}
const mathSpan: Construct = {
name: 'mathSpanSingleDollar',
tokenize: tokenizeMathSpan,
previous,
};
/**
* micromark syntax extension adding currency-safe single-dollar inline math.
* It emits the same `mathText`/`mathTextData` tokens as
* `micromark-extension-math`, so `remark-math`'s `mathFromMarkdown` handlers
* turn its spans into regular `inlineMath` nodes. Registration order relative
* to `remark-math` is immaterial: this construct rejects `$$` openers and
* `remark-math` (with `singleDollarTextMath: false`) rejects single `$` openers.
*/
export const singleDollarMath: Extension = {
text: { [codes.dollarSign]: mathSpan },
};
/** remark plugin for {@link singleDollarMath}; `remark-math` must run alongside it. */
export function remarkSingleDollarMath(this: Processor) {
addMicromarkExtension(this, singleDollarMath);
}
@@ -0,0 +1,398 @@
/**
* LaTeX's own math delimiters, ``\( … \)`` and ``\[ … \]``, as micromark
* constructs. Models write math this way far more often than with dollars, and
* remark-math only knows dollars.
*
* The constructs emit the same tokens as `micromark-extension-math`, so
* `remark-math`'s `mathFromMarkdown` turns them into ordinary `inlineMath` and
* `math` nodes and rehype-katex renders them. `remark-math` has to run too.
*
* - Text: ``\(x\)`` and ``\[x\]`` anywhere in a paragraph become inline math.
* The span may wrap lines but not cross a paragraph; ``\\`` pairs are opaque
* (a LaTeX line break), a nested ``\( … \)`` of the same kind is kept whole,
* and whitespace-only content is not math, so an escaped task box
* ``- \[ \] todo`` stays text. ``\[`` right after a letter or digit is an
* escaped index (``arr\[0\]``), not math.
* - Flow: a ``\[`` that starts a line opens display math, closed by a ``\]``
* that ends a line (the same line or a later one). Its lines are not parsed
* as markdown, so a lone ``=`` is not a setext underline. If a ``\]`` is
* followed by more text on its line, or the block is never closed, it is not
* a display block, and the text construct gets the ``\[`` instead: one line
* such as ``\[a\] and more`` must not swallow the rest of the answer.
*
* Written for this app rather than aliasing `micromark-extension-llm-math`
* over `micromark-extension-math`, whose flow construct does exactly that
* swallowing. Its structure follows that package and `micromark-extension-math`
* (both MIT).
*/
import {
asciiAlphanumeric,
markdownLineEnding,
markdownSpace,
} from 'micromark-util-character';
import { codes, types } from 'micromark-util-symbol';
import type {
Code,
Construct,
Effects,
Extension,
State,
TokenizeContext,
} from 'micromark-util-types';
import type { Processor } from 'unified';
import { addMicromarkExtension } from './micromarkExtension';
function tokenizeTexText(
this: TokenizeContext,
effects: Effects,
ok: State,
nok: State,
): State {
// What precedes the opening backslash, read before it is consumed.
const before = this.previous;
let open: Code = null;
let close: Code = null;
let depth = 0;
let hasContent = false;
const closing: Construct = { tokenize: tokenizeClosing, partial: true };
return start;
function start(code: Code): State | undefined {
effects.enter('mathText');
effects.enter('mathTextSequence');
effects.consume(code);
return sequenceOpen;
}
function sequenceOpen(code: Code): State | undefined {
if (code === codes.leftParenthesis) {
close = codes.rightParenthesis;
} else if (code === codes.leftSquareBracket && !asciiAlphanumeric(before)) {
close = codes.rightSquareBracket;
} else {
return nok(code);
}
open = code;
effects.consume(code);
effects.exit('mathTextSequence');
return between;
}
function between(code: Code): State | undefined {
if (code === codes.eof) return nok(code);
if (code === codes.backslash) {
return effects.check(closing, closeSequence, escape)(code);
}
// A line ending is kept as data: TeX reads it as a space, and dropping it
// would fuse ``\alpha`` with the next line's first letter.
if (markdownLineEnding(code)) {
effects.enter('mathTextData');
effects.consume(code);
effects.exit('mathTextData');
return between;
}
effects.enter('mathTextData');
return data(code);
}
function data(code: Code): State | undefined {
if (
code === codes.eof ||
code === codes.backslash ||
markdownLineEnding(code)
) {
effects.exit('mathTextData');
return between(code);
}
if (!markdownSpace(code)) hasContent = true;
effects.consume(code);
return data;
}
function escape(code: Code): State | undefined {
effects.enter('mathTextData');
effects.consume(code);
hasContent = true;
return escaped;
}
function escaped(code: Code): State | undefined {
if (code === codes.eof || markdownLineEnding(code)) {
effects.exit('mathTextData');
return between(code);
}
if (code === open) depth++;
// Only reached for a nested close: the outermost one is `closing`.
if (code === close) depth--;
effects.consume(code);
return data;
}
function closeSequence(code: Code): State | undefined {
if (!hasContent) return nok(code);
effects.enter('mathTextSequence');
effects.consume(code);
return closeBracket;
}
function closeBracket(code: Code): State | undefined {
effects.consume(code);
return done;
}
function done(code: Code): State | undefined {
effects.exit('mathTextSequence');
effects.exit('mathText');
return ok(code);
}
function tokenizeClosing(effects: Effects, ok: State, nok: State): State {
return (code) => {
effects.enter('mathTextSequence');
effects.consume(code);
return (next) => {
if (next !== close || depth !== 0) return nok(next);
effects.exit('mathTextSequence');
return ok(next);
};
};
}
}
/** A ``\(`` or ``\[`` opens math unless its backslash is itself escaped. */
function previousTex(this: TokenizeContext, code: Code): boolean {
return (
code !== codes.backslash ||
this.events[this.events.length - 1][1].type === types.characterEscape
);
}
const texText: Construct = {
name: 'mathTexText',
tokenize: tokenizeTexText,
previous: previousTex,
};
function tokenizeTexFlow(
this: TokenizeContext,
effects: Effects,
ok: State,
nok: State,
): State {
// Set while micromark asks whether the block may interrupt a paragraph.
const interrupt = this.interrupt;
// Whether the line being entered is a lazy continuation.
const lazyLine = () => this.parser.lazy[this.now().line];
const tail = this.events[this.events.length - 1];
// The fence's own indent, stripped from the lines inside it too.
const initialSize =
tail && tail[1].type === types.linePrefix
? tail[2].sliceSerialize(tail[1], true).length
: 0;
const closingFence: Construct = {
tokenize: tokenizeClosingFence,
partial: true,
};
const nonLazyLine: Construct = {
tokenize: tokenizeNonLazyLine,
partial: true,
};
return start;
function start(code: Code): State | undefined {
effects.enter('mathFlow');
effects.enter('mathFlowFence');
effects.enter('mathFlowFenceSequence');
effects.consume(code);
return sequenceOpen;
}
function sequenceOpen(code: Code): State | undefined {
if (code !== codes.leftSquareBracket) return nok(code);
effects.consume(code);
effects.exit('mathFlowFenceSequence');
return fenceWhitespace;
}
function fenceWhitespace(code: Code): State | undefined {
if (!markdownSpace(code)) return fenceEnd(code);
effects.enter(types.whitespace);
return fenceWhitespaceInside(code);
}
function fenceWhitespaceInside(code: Code): State | undefined {
if (markdownSpace(code)) {
effects.consume(code);
return fenceWhitespaceInside;
}
effects.exit(types.whitespace);
return fenceEnd(code);
}
function fenceEnd(code: Code): State | undefined {
effects.exit('mathFlowFence');
return lineContent(code);
}
// Anywhere in a line of the block, outside a value chunk.
function lineContent(code: Code): State | undefined {
if (code === codes.eof) return nok(code);
if (markdownLineEnding(code)) {
// To interrupt a paragraph only the opening line has to hold up.
if (interrupt) return ok(code);
return effects.attempt(nonLazyLine, lineStart, nok)(code);
}
if (code === codes.backslash) {
return effects.check(
{ tokenize: tokenizeBracketClose, partial: true },
atClose,
valueEscape,
)(code);
}
effects.enter('mathFlowValue');
return value(code);
}
function lineStart(code: Code): State | undefined {
if (initialSize === 0 || !markdownSpace(code)) return lineContent(code);
effects.enter(types.linePrefix);
return linePrefix(code, 0);
}
function linePrefix(code: Code, size: number): State | undefined {
if (markdownSpace(code) && size < initialSize) {
effects.consume(code);
return (next: Code) => linePrefix(next, size + 1);
}
effects.exit(types.linePrefix);
return lineContent(code);
}
function value(code: Code): State | undefined {
if (
code === codes.eof ||
code === codes.backslash ||
markdownLineEnding(code)
) {
effects.exit('mathFlowValue');
return lineContent(code);
}
effects.consume(code);
return value;
}
// A ``\]`` closes the block only when the rest of its line is blank; with
// anything after it, this is a line of prose, not a display block.
function atClose(code: Code): State | undefined {
return effects.attempt(closingFence, after, nok)(code);
}
function after(code: Code): State | undefined {
effects.exit('mathFlow');
return ok(code);
}
// ``\\`` and ``\x`` pairs are content, so ``\\]`` is not a close.
function valueEscape(code: Code): State | undefined {
effects.enter('mathFlowValue');
effects.consume(code);
return valueEscaped;
}
function valueEscaped(code: Code): State | undefined {
if (code === codes.eof || markdownLineEnding(code)) {
effects.exit('mathFlowValue');
return lineContent(code);
}
effects.consume(code);
return value;
}
function tokenizeBracketClose(
effects: Effects,
ok: State,
nok: State,
): State {
return (code) => {
effects.enter('mathFlowFenceSequence');
effects.consume(code);
return (next) => {
if (next !== codes.rightSquareBracket) return nok(next);
effects.exit('mathFlowFenceSequence');
return ok(next);
};
};
}
function tokenizeClosingFence(
effects: Effects,
ok: State,
nok: State,
): State {
return (code) => {
effects.enter('mathFlowFence');
effects.enter('mathFlowFenceSequence');
effects.consume(code);
return (bracket) => {
effects.consume(bracket);
effects.exit('mathFlowFenceSequence');
return afterSequence;
};
};
function afterSequence(code: Code): State | undefined {
if (markdownSpace(code)) {
effects.enter(types.whitespace);
return trailing(code);
}
return end(code);
}
function trailing(code: Code): State | undefined {
if (markdownSpace(code)) {
effects.consume(code);
return trailing;
}
effects.exit(types.whitespace);
return end(code);
}
function end(code: Code): State | undefined {
if (code !== codes.eof && !markdownLineEnding(code)) return nok(code);
effects.exit('mathFlowFence');
return ok(code);
}
}
// Past a line ending, unless the next line is a lazy continuation, which
// means the container around the block (a quote, a list item) has ended.
function tokenizeNonLazyLine(effects: Effects, ok: State, nok: State): State {
return (code) => {
effects.enter(types.lineEnding);
effects.consume(code);
effects.exit(types.lineEnding);
return (next) => (lazyLine() ? nok(next) : ok(next));
};
}
}
const texFlow: Construct = {
name: 'mathTexFlow',
tokenize: tokenizeTexFlow,
concrete: true,
};
export const texMath: Extension = {
flow: { [codes.backslash]: texFlow },
text: { [codes.backslash]: texText },
};
/** remark plugin for {@link texMath}; `remark-math` must run alongside it. */
export function remarkTexMath(this: Processor) {
addMicromarkExtension(this, texMath);
}