diff --git a/frontend/src/conversation/MarkdownAnswer.codeSpans.test.ts b/frontend/src/conversation/MarkdownAnswer.codeSpans.test.ts
new file mode 100644
index 00000000..dc3e0be9
--- /dev/null
+++ b/frontend/src/conversation/MarkdownAnswer.codeSpans.test.ts
@@ -0,0 +1,166 @@
+/**
+ * Code must survive the markdown string pre-passes.
+ *
+ * ``processMarkdownContent`` rewrote every ``[N]`` as ``[N](#cite-N)`` and
+ * ``preprocessLaTeX`` every ``\[…\]`` as ``$$…$$``, neither with any awareness
+ * of code, so an index subscript or a bash prompt escape came out corrupted in
+ * the rendered answer — and in what the copy button copied.
+ *
+ * Neither rewrite can move into a remark plugin, where a ``code`` node would be
+ * distinct from a ``text`` one: remark-math tokenises ``$``/``$$`` while
+ * parsing, so the LaTeX pass has to happen on the raw string. Hence the shared
+ * ``applyOutsideCode`` masking helper rather than an AST visitor.
+ */
+import { describe, expect, it } from 'vitest';
+
+import {
+ applyOutsideCode,
+ preprocessLaTeX,
+ processMarkdownContent,
+} from './MarkdownAnswer';
+
+/** The rendered text of a single-segment answer. */
+const text = (source: string) => processMarkdownContent(source)[0].content;
+
+describe('citation rewriting spares code', () => {
+ it('leaves an index subscript alone inside a fence', () => {
+ const source = '```python\nfor row in cur:\n print(row[0])\n```';
+ expect(text(source)).toBe(source);
+ });
+
+ it('leaves an index subscript alone inside an inline span', () => {
+ expect(text('Use `items[2]` here.')).toBe('Use `items[2]` here.');
+ });
+
+ it('leaves a mid-sentence triple-backtick span alone', () => {
+ expect(text('inline ```code[0]``` end')).toBe('inline ```code[0]``` end');
+ });
+
+ it('leaves an inline span that wraps a line alone', () => {
+ expect(text('use `arr[0]\nnext` here')).toBe('use `arr[0]\nnext` here');
+ });
+
+ it('leaves a tilde fence alone', () => {
+ const source = '~~~python\nprint(row[0])\n~~~';
+ expect(text(source)).toBe(source);
+ });
+
+ it('leaves a fence nested in a longer one alone', () => {
+ // A ``` run does not close a ```` fence, so the inner block is content.
+ const source = '````markdown\n```js\nconst a = arr[0];\n```\n````';
+ expect(text(source)).toBe(source);
+ });
+
+ it('leaves a fence indented under a list item alone', () => {
+ const source =
+ '1. Step:\n\n ```js\n const a = arr[0];\n\n const b = arr[1];\n ```';
+ expect(text(source)).toBe(source);
+ });
+
+ it('leaves an unterminated fence alone while the answer is still streaming', () => {
+ const source = '```python\nprint(row[0]';
+ expect(text(source)).toBe(source);
+ });
+
+ it('leaves an unterminated inline span alone while streaming', () => {
+ expect(text('Use `items[2')).toBe('Use `items[2');
+ });
+
+ it('leaves mermaid node labels alone', () => {
+ // The cite pass runs before the ```mermaid split, so diagram source was
+ // corrupted too and the render failed.
+ const source = '```mermaid\nflowchart TD\n A[0] --> B[1]\n```';
+ expect(processMarkdownContent(source)[0].content).toBe(
+ 'flowchart TD\n A[0] --> B[1]',
+ );
+ });
+
+ it('still links a real citation in prose', () => {
+ expect(text('See source [1] for details.')).toBe(
+ 'See source [1](#cite-1) for details.',
+ );
+ });
+
+ it('links prose citations on either side of a fence without touching it', () => {
+ expect(
+ text('Per [1], run:\n```js\nconst a = arr[0];\n```\nthen [2].'),
+ ).toBe(
+ 'Per [1](#cite-1), run:\n```js\nconst a = arr[0];\n```\nthen [2](#cite-2).',
+ );
+ });
+
+ it('still links a citation on an indented list continuation line', () => {
+ // Four-space indents are list content far more often than they are code,
+ // so they are prose to the masker.
+ expect(text('- one\n - nested [1]')).toBe(
+ '- one\n - nested [1](#cite-1)',
+ );
+ });
+
+ it('leaves an already-linked reference alone', () => {
+ expect(text('see [1](#cite-1) ok')).toBe('see [1](#cite-1) ok');
+ });
+});
+
+describe('LaTeX preprocessing spares code', () => {
+ it('leaves bash prompt escapes alone', () => {
+ // `\[`/`\]` are non-printing-sequence markers in PS1; the block rule turned
+ // them into `$$` delimiters and broke the prompt.
+ const source = '```bash\nPS1="\\[\\e[0m\\]$ "\n```';
+ expect(preprocessLaTeX(source)).toBe(source);
+ });
+
+ it('leaves bracket character classes alone in an inline span', () => {
+ expect(preprocessLaTeX('match `\\[0-9\\]` here')).toBe(
+ 'match `\\[0-9\\]` here',
+ );
+ });
+
+ it('still converts real inline math in prose', () => {
+ expect(preprocessLaTeX('The value \\(x^2\\) is fine.')).toBe(
+ 'The value $x^2$ is fine.',
+ );
+ });
+
+ it('still converts real block math in prose', () => {
+ expect(preprocessLaTeX('Given \\[a+b\\] we get')).toBe(
+ 'Given $$a+b$$ we get',
+ );
+ });
+
+ it('does not link a subscript inside the math it just produced', () => {
+ // `$a[1](#cite-1)$` is not something KaTeX can render.
+ expect(text('value \\(a[1]\\) end')).toBe('value $a[1]$ end');
+ expect(text('given \\[a[1]+b\\] we get [2]')).toBe(
+ 'given $$a[1]+b$$ we get [2](#cite-2)',
+ );
+ });
+});
+
+describe('applyOutsideCode', () => {
+ const shout = (segment: string) => segment.toUpperCase();
+
+ it('transforms prose either side of every code span', () => {
+ expect(applyOutsideCode('a `b` c ```d``` e', shout)).toBe(
+ 'A `b` C ```d``` E',
+ );
+ });
+
+ it('handles several inline spans on one line', () => {
+ expect(applyOutsideCode('x `a` y `b` z', shout)).toBe('X `a` Y `b` Z');
+ });
+
+ it('treats a double-backtick span containing a backtick as one span', () => {
+ expect(applyOutsideCode('x ``a`b`` y', shout)).toBe('X ``a`b`` Y');
+ });
+
+ it('does not close a long fence on a shorter run', () => {
+ expect(applyOutsideCode('````\na\n```\nb\n````\nc', shout)).toBe(
+ '````\na\n```\nb\n````\nC',
+ );
+ });
+
+ it('passes an empty string through', () => {
+ expect(applyOutsideCode('', shout)).toBe('');
+ });
+});
diff --git a/frontend/src/conversation/MarkdownAnswer.tsx b/frontend/src/conversation/MarkdownAnswer.tsx
index 903b7a98..479444a8 100644
--- a/frontend/src/conversation/MarkdownAnswer.tsx
+++ b/frontend/src/conversation/MarkdownAnswer.tsx
@@ -22,30 +22,87 @@ import {
sandboxUrlTransform,
} from './sandboxLinks';
+// One fenced block or inline code span. Backtick runs are length-matched, so a
+// ```` fence closes only on ```` and nested fences stay masked. The
+// unterminated alternatives keep an open span matched too: answers stream in,
+// so a fence is open for most of its life and needs protecting the whole time,
+// not just once the closing fence arrives. Four-space indented blocks are
+// deliberately not masked — list continuation lines are indented the same way,
+// and masking those would drop real citations out of nested lists.
+const CODE_SPAN =
+ /(? string,
+ onMasked: (segment: string) => string = (segment) => segment,
+): string {
+ const pattern = new RegExp(mask.source, mask.flags); // its own lastIndex
+ let result = '';
+ let index = 0;
+ let match: RegExpExecArray | null;
+
+ while ((match = pattern.exec(content)) !== null) {
+ result += transform(content.slice(index, match.index)) + onMasked(match[0]);
+ index = pattern.lastIndex;
+ }
+ return result + transform(content.slice(index));
+}
+
+// Runs `transform` over the prose in `content`, leaving code untouched.
+//
+// The rewrites below are plain regex over the whole document, and both used to
+// corrupt the user's own code: `print(row[0])` rendered as
+// `print(row[0](#cite-0))`, and `PS1="\[\e[0m\]"` as `PS1="$$\e[0m$$"`. Neither
+// can move into a remark plugin: remark-math tokenises `$`/`$$` while parsing,
+// so the LaTeX pass has to run on the raw string before remark ever sees it.
+export function applyOutsideCode(
+ content: string,
+ transform: (segment: string) => string,
+): string {
+ return transformOutside(content, CODE_SPAN, transform);
+}
+
+// Rewrites one LaTeX span into the ``$$``/``$`` form remark-math parses.
+function toDollarMath(span: string): string {
+ const equation = span.slice(2, -2);
+ return span.startsWith('\\[') ? `$$${equation}$$` : `$${equation}$`;
+}
+
// Replaces block-level ``\[ \]`` and inline ``\( \)`` LaTeX delimiters with the
// ``$$``/``$`` forms remark-math understands.
export function preprocessLaTeX(content: string): string {
- const blockProcessedContent = content.replace(
- /\\\[(.*?)\\\]/gs,
- (_, equation) => `$$${equation}$$`,
+ return applyOutsideCode(content, (prose) =>
+ prose.replace(LATEX_SPAN, toDollarMath),
);
- return blockProcessedContent.replace(
- /\\\((.*?)\\\)/gs,
- (_, equation) => `$${equation}$`,
+}
+
+// Turns citation references ``[N]`` into ``[N](#cite-N)`` links so
+// ReactMarkdown renders them as tags we can style. The lookarounds skip
+// references that are already links.
+function linkCitations(prose: string): string {
+ return prose.replace(
+ /(? `[${num}](#cite-${num})`,
);
}
type ContentSegment = { type: 'text' | 'mermaid'; content: string };
export function processMarkdownContent(content: string): ContentSegment[] {
- let processedContent = preprocessLaTeX(content);
-
- // Convert citation references [N] into markdown links [N](#cite-N)
- // so ReactMarkdown renders them as tags we can style.
- // Avoid matching inside code blocks or existing links.
- processedContent = processedContent.replace(
- /(? `[${num}](#cite-${num})`,
+ // Citations are linked outside code — an index subscript like `row[0]` is not
+ // a citation — and outside the math this same pass produces. Math the answer
+ // already wrote as ``$…$`` stays unmasked, since ``$`` is also a currency
+ // sign. This has to run before the ```mermaid split below, so diagram source
+ // needs the same guard.
+ const processedContent = applyOutsideCode(content, (prose) =>
+ transformOutside(prose, LATEX_SPAN, linkCitations, toDollarMath),
);
const contentSegments: ContentSegment[] = [];