diff --git a/frontend/src/conversation/MarkdownAnswer.codeSpans.test.ts b/frontend/src/conversation/MarkdownAnswer.codeSpans.test.ts new file mode 100644 index 00000000..dc3e0be9 --- /dev/null +++ b/frontend/src/conversation/MarkdownAnswer.codeSpans.test.ts @@ -0,0 +1,166 @@ +/** + * Code must survive the markdown string pre-passes. + * + * ``processMarkdownContent`` rewrote every ``[N]`` as ``[N](#cite-N)`` and + * ``preprocessLaTeX`` every ``\[…\]`` as ``$$…$$``, neither with any awareness + * of code, so an index subscript or a bash prompt escape came out corrupted in + * the rendered answer — and in what the copy button copied. + * + * Neither rewrite can move into a remark plugin, where a ``code`` node would be + * distinct from a ``text`` one: remark-math tokenises ``$``/``$$`` while + * parsing, so the LaTeX pass has to happen on the raw string. Hence the shared + * ``applyOutsideCode`` masking helper rather than an AST visitor. + */ +import { describe, expect, it } from 'vitest'; + +import { + applyOutsideCode, + preprocessLaTeX, + processMarkdownContent, +} from './MarkdownAnswer'; + +/** The rendered text of a single-segment answer. */ +const text = (source: string) => processMarkdownContent(source)[0].content; + +describe('citation rewriting spares code', () => { + it('leaves an index subscript alone inside a fence', () => { + const source = '```python\nfor row in cur:\n print(row[0])\n```'; + expect(text(source)).toBe(source); + }); + + it('leaves an index subscript alone inside an inline span', () => { + expect(text('Use `items[2]` here.')).toBe('Use `items[2]` here.'); + }); + + it('leaves a mid-sentence triple-backtick span alone', () => { + expect(text('inline ```code[0]``` end')).toBe('inline ```code[0]``` end'); + }); + + it('leaves an inline span that wraps a line alone', () => { + expect(text('use `arr[0]\nnext` here')).toBe('use `arr[0]\nnext` here'); + }); + + it('leaves a tilde fence alone', () => { + const source = '~~~python\nprint(row[0])\n~~~'; + expect(text(source)).toBe(source); + }); + + it('leaves a fence nested in a longer one alone', () => { + // A ``` run does not close a ```` fence, so the inner block is content. + const source = '````markdown\n```js\nconst a = arr[0];\n```\n````'; + expect(text(source)).toBe(source); + }); + + it('leaves a fence indented under a list item alone', () => { + const source = + '1. Step:\n\n ```js\n const a = arr[0];\n\n const b = arr[1];\n ```'; + expect(text(source)).toBe(source); + }); + + it('leaves an unterminated fence alone while the answer is still streaming', () => { + const source = '```python\nprint(row[0]'; + expect(text(source)).toBe(source); + }); + + it('leaves an unterminated inline span alone while streaming', () => { + expect(text('Use `items[2')).toBe('Use `items[2'); + }); + + it('leaves mermaid node labels alone', () => { + // The cite pass runs before the ```mermaid split, so diagram source was + // corrupted too and the render failed. + const source = '```mermaid\nflowchart TD\n A[0] --> B[1]\n```'; + expect(processMarkdownContent(source)[0].content).toBe( + 'flowchart TD\n A[0] --> B[1]', + ); + }); + + it('still links a real citation in prose', () => { + expect(text('See source [1] for details.')).toBe( + 'See source [1](#cite-1) for details.', + ); + }); + + it('links prose citations on either side of a fence without touching it', () => { + expect( + text('Per [1], run:\n```js\nconst a = arr[0];\n```\nthen [2].'), + ).toBe( + 'Per [1](#cite-1), run:\n```js\nconst a = arr[0];\n```\nthen [2](#cite-2).', + ); + }); + + it('still links a citation on an indented list continuation line', () => { + // Four-space indents are list content far more often than they are code, + // so they are prose to the masker. + expect(text('- one\n - nested [1]')).toBe( + '- one\n - nested [1](#cite-1)', + ); + }); + + it('leaves an already-linked reference alone', () => { + expect(text('see [1](#cite-1) ok')).toBe('see [1](#cite-1) ok'); + }); +}); + +describe('LaTeX preprocessing spares code', () => { + it('leaves bash prompt escapes alone', () => { + // `\[`/`\]` are non-printing-sequence markers in PS1; the block rule turned + // them into `$$` delimiters and broke the prompt. + const source = '```bash\nPS1="\\[\\e[0m\\]$ "\n```'; + expect(preprocessLaTeX(source)).toBe(source); + }); + + it('leaves bracket character classes alone in an inline span', () => { + expect(preprocessLaTeX('match `\\[0-9\\]` here')).toBe( + 'match `\\[0-9\\]` here', + ); + }); + + it('still converts real inline math in prose', () => { + expect(preprocessLaTeX('The value \\(x^2\\) is fine.')).toBe( + 'The value $x^2$ is fine.', + ); + }); + + it('still converts real block math in prose', () => { + expect(preprocessLaTeX('Given \\[a+b\\] we get')).toBe( + 'Given $$a+b$$ we get', + ); + }); + + it('does not link a subscript inside the math it just produced', () => { + // `$a[1](#cite-1)$` is not something KaTeX can render. + expect(text('value \\(a[1]\\) end')).toBe('value $a[1]$ end'); + expect(text('given \\[a[1]+b\\] we get [2]')).toBe( + 'given $$a[1]+b$$ we get [2](#cite-2)', + ); + }); +}); + +describe('applyOutsideCode', () => { + const shout = (segment: string) => segment.toUpperCase(); + + it('transforms prose either side of every code span', () => { + expect(applyOutsideCode('a `b` c ```d``` e', shout)).toBe( + 'A `b` C ```d``` E', + ); + }); + + it('handles several inline spans on one line', () => { + expect(applyOutsideCode('x `a` y `b` z', shout)).toBe('X `a` Y `b` Z'); + }); + + it('treats a double-backtick span containing a backtick as one span', () => { + expect(applyOutsideCode('x ``a`b`` y', shout)).toBe('X ``a`b`` Y'); + }); + + it('does not close a long fence on a shorter run', () => { + expect(applyOutsideCode('````\na\n```\nb\n````\nc', shout)).toBe( + '````\na\n```\nb\n````\nC', + ); + }); + + it('passes an empty string through', () => { + expect(applyOutsideCode('', shout)).toBe(''); + }); +}); diff --git a/frontend/src/conversation/MarkdownAnswer.tsx b/frontend/src/conversation/MarkdownAnswer.tsx index 903b7a98..479444a8 100644 --- a/frontend/src/conversation/MarkdownAnswer.tsx +++ b/frontend/src/conversation/MarkdownAnswer.tsx @@ -22,30 +22,87 @@ import { sandboxUrlTransform, } from './sandboxLinks'; +// One fenced block or inline code span. Backtick runs are length-matched, so a +// ```` fence closes only on ```` and nested fences stay masked. The +// unterminated alternatives keep an open span matched too: answers stream in, +// so a fence is open for most of its life and needs protecting the whole time, +// not just once the closing fence arrives. Four-space indented blocks are +// deliberately not masked — list continuation lines are indented the same way, +// and masking those would drop real citations out of nested lists. +const CODE_SPAN = + /(? string, + onMasked: (segment: string) => string = (segment) => segment, +): string { + const pattern = new RegExp(mask.source, mask.flags); // its own lastIndex + let result = ''; + let index = 0; + let match: RegExpExecArray | null; + + while ((match = pattern.exec(content)) !== null) { + result += transform(content.slice(index, match.index)) + onMasked(match[0]); + index = pattern.lastIndex; + } + return result + transform(content.slice(index)); +} + +// Runs `transform` over the prose in `content`, leaving code untouched. +// +// The rewrites below are plain regex over the whole document, and both used to +// corrupt the user's own code: `print(row[0])` rendered as +// `print(row[0](#cite-0))`, and `PS1="\[\e[0m\]"` as `PS1="$$\e[0m$$"`. Neither +// can move into a remark plugin: remark-math tokenises `$`/`$$` while parsing, +// so the LaTeX pass has to run on the raw string before remark ever sees it. +export function applyOutsideCode( + content: string, + transform: (segment: string) => string, +): string { + return transformOutside(content, CODE_SPAN, transform); +} + +// Rewrites one LaTeX span into the ``$$``/``$`` form remark-math parses. +function toDollarMath(span: string): string { + const equation = span.slice(2, -2); + return span.startsWith('\\[') ? `$$${equation}$$` : `$${equation}$`; +} + // Replaces block-level ``\[ \]`` and inline ``\( \)`` LaTeX delimiters with the // ``$$``/``$`` forms remark-math understands. export function preprocessLaTeX(content: string): string { - const blockProcessedContent = content.replace( - /\\\[(.*?)\\\]/gs, - (_, equation) => `$$${equation}$$`, + return applyOutsideCode(content, (prose) => + prose.replace(LATEX_SPAN, toDollarMath), ); - return blockProcessedContent.replace( - /\\\((.*?)\\\)/gs, - (_, equation) => `$${equation}$`, +} + +// Turns citation references ``[N]`` into ``[N](#cite-N)`` links so +// ReactMarkdown renders them as tags we can style. The lookarounds skip +// references that are already links. +function linkCitations(prose: string): string { + return prose.replace( + /(? `[${num}](#cite-${num})`, ); } type ContentSegment = { type: 'text' | 'mermaid'; content: string }; export function processMarkdownContent(content: string): ContentSegment[] { - let processedContent = preprocessLaTeX(content); - - // Convert citation references [N] into markdown links [N](#cite-N) - // so ReactMarkdown renders them as tags we can style. - // Avoid matching inside code blocks or existing links. - processedContent = processedContent.replace( - /(? `[${num}](#cite-${num})`, + // Citations are linked outside code — an index subscript like `row[0]` is not + // a citation — and outside the math this same pass produces. Math the answer + // already wrote as ``$…$`` stays unmasked, since ``$`` is also a currency + // sign. This has to run before the ```mermaid split below, so diagram source + // needs the same guard. + const processedContent = applyOutsideCode(content, (prose) => + transformOutside(prose, LATEX_SPAN, linkCitations, toDollarMath), ); const contentSegments: ContentSegment[] = [];