fix: md citation issue

This commit is contained in:
Alex committed 2026-08-25 00:15:12 +01:00
1 parent 5c55d2610b
commit 2ff895ac9c
2 files changed
+237 -14

No files matched your search

@@ -0,0 +1,166 @@
/**
* Code must survive the markdown string pre-passes.
*
* ``processMarkdownContent`` rewrote every ``[N]`` as ``[N](#cite-N)`` and
* ``preprocessLaTeX`` every ``\[…\]`` as ``$$…$$``, neither with any awareness
* of code, so an index subscript or a bash prompt escape came out corrupted in
* the rendered answer — and in what the copy button copied.
*
* Neither rewrite can move into a remark plugin, where a ``code`` node would be
* distinct from a ``text`` one: remark-math tokenises ``$``/``$$`` while
* parsing, so the LaTeX pass has to happen on the raw string. Hence the shared
* ``applyOutsideCode`` masking helper rather than an AST visitor.
*/
import { describe, expect, it } from 'vitest';
import {
applyOutsideCode,
preprocessLaTeX,
processMarkdownContent,
} from './MarkdownAnswer';
/** The rendered text of a single-segment answer. */
const text = (source: string) => processMarkdownContent(source)[0].content;
describe('citation rewriting spares code', () => {
it('leaves an index subscript alone inside a fence', () => {
const source = '```python\nfor row in cur:\n print(row[0])\n```';
expect(text(source)).toBe(source);
});
it('leaves an index subscript alone inside an inline span', () => {
expect(text('Use `items[2]` here.')).toBe('Use `items[2]` here.');
});
it('leaves a mid-sentence triple-backtick span alone', () => {
expect(text('inline ```code[0]``` end')).toBe('inline ```code[0]``` end');
});
it('leaves an inline span that wraps a line alone', () => {
expect(text('use `arr[0]\nnext` here')).toBe('use `arr[0]\nnext` here');
});
it('leaves a tilde fence alone', () => {
const source = '~~~python\nprint(row[0])\n~~~';
expect(text(source)).toBe(source);
});
it('leaves a fence nested in a longer one alone', () => {
// A ``` run does not close a ```` fence, so the inner block is content.
const source = '````markdown\n```js\nconst a = arr[0];\n```\n````';
expect(text(source)).toBe(source);
});
it('leaves a fence indented under a list item alone', () => {
const source =
'1. Step:\n\n ```js\n const a = arr[0];\n\n const b = arr[1];\n ```';
expect(text(source)).toBe(source);
});
it('leaves an unterminated fence alone while the answer is still streaming', () => {
const source = '```python\nprint(row[0]';
expect(text(source)).toBe(source);
});
it('leaves an unterminated inline span alone while streaming', () => {
expect(text('Use `items[2')).toBe('Use `items[2');
});
it('leaves mermaid node labels alone', () => {
// The cite pass runs before the ```mermaid split, so diagram source was
// corrupted too and the render failed.
const source = '```mermaid\nflowchart TD\n A[0] --> B[1]\n```';
expect(processMarkdownContent(source)[0].content).toBe(
'flowchart TD\n A[0] --> B[1]',
);
});
it('still links a real citation in prose', () => {
expect(text('See source [1] for details.')).toBe(
'See source [1](#cite-1) for details.',
);
});
it('links prose citations on either side of a fence without touching it', () => {
expect(
text('Per [1], run:\n```js\nconst a = arr[0];\n```\nthen [2].'),
).toBe(
'Per [1](#cite-1), run:\n```js\nconst a = arr[0];\n```\nthen [2](#cite-2).',
);
});
it('still links a citation on an indented list continuation line', () => {
// Four-space indents are list content far more often than they are code,
// so they are prose to the masker.
expect(text('- one\n - nested [1]')).toBe(
'- one\n - nested [1](#cite-1)',
);
});
it('leaves an already-linked reference alone', () => {
expect(text('see [1](#cite-1) ok')).toBe('see [1](#cite-1) ok');
});
});
describe('LaTeX preprocessing spares code', () => {
it('leaves bash prompt escapes alone', () => {
// `\[`/`\]` are non-printing-sequence markers in PS1; the block rule turned
// them into `$$` delimiters and broke the prompt.
const source = '```bash\nPS1="\\[\\e[0m\\]$ "\n```';
expect(preprocessLaTeX(source)).toBe(source);
});
it('leaves bracket character classes alone in an inline span', () => {
expect(preprocessLaTeX('match `\\[0-9\\]` here')).toBe(
'match `\\[0-9\\]` here',
);
});
it('still converts real inline math in prose', () => {
expect(preprocessLaTeX('The value \\(x^2\\) is fine.')).toBe(
'The value $x^2$ is fine.',
);
});
it('still converts real block math in prose', () => {
expect(preprocessLaTeX('Given \\[a+b\\] we get')).toBe(
'Given $$a+b$$ we get',
);
});
it('does not link a subscript inside the math it just produced', () => {
// `$a[1](#cite-1)$` is not something KaTeX can render.
expect(text('value \\(a[1]\\) end')).toBe('value $a[1]$ end');
expect(text('given \\[a[1]+b\\] we get [2]')).toBe(
'given $$a[1]+b$$ we get [2](#cite-2)',
);
});
});
describe('applyOutsideCode', () => {
const shout = (segment: string) => segment.toUpperCase();
it('transforms prose either side of every code span', () => {
expect(applyOutsideCode('a `b` c ```d``` e', shout)).toBe(
'A `b` C ```d``` E',
);
});
it('handles several inline spans on one line', () => {
expect(applyOutsideCode('x `a` y `b` z', shout)).toBe('X `a` Y `b` Z');
});
it('treats a double-backtick span containing a backtick as one span', () => {
expect(applyOutsideCode('x ``a`b`` y', shout)).toBe('X ``a`b`` Y');
});
it('does not close a long fence on a shorter run', () => {
expect(applyOutsideCode('````\na\n```\nb\n````\nc', shout)).toBe(
'````\na\n```\nb\n````\nC',
);
});
it('passes an empty string through', () => {
expect(applyOutsideCode('', shout)).toBe('');
});
});
+71 -14
View File
@@ -22,30 +22,87 @@ import {
sandboxUrlTransform,
} from './sandboxLinks';
// One fenced block or inline code span. Backtick runs are length-matched, so a
// ```` fence closes only on ```` and nested fences stay masked. The
// unterminated alternatives keep an open span matched too: answers stream in,
// so a fence is open for most of its life and needs protecting the whole time,
// not just once the closing fence arrives. Four-space indented blocks are
// deliberately not masked — list continuation lines are indented the same way,
// and masking those would drop real citations out of nested lists.
const CODE_SPAN =
/(?<![^\n])[ \t]*(`{3,})[^`\n]*(?:[\s\S]*?\n[ \t]*\1`*[ \t]*(?![^\n])|[\s\S]*$)|(?<![^\n])[ \t]*(~{3,})[^\n]*(?:[\s\S]*?\n[ \t]*\2~*[ \t]*(?![^\n])|[\s\S]*$)|(`+)(?:[^\n]|\n(?![ \t]*\n))*?\3(?!`)|`+[^`\n]*$/g;
// ``\[ \]`` and ``\( \)`` LaTeX, which remark-math does not recognise.
const LATEX_SPAN = /\\\[[\s\S]*?\\\]|\\\([\s\S]*?\\\)/g;
// Applies `transform` to everything outside the regions `mask` matches, and
// `onMasked` to those regions themselves.
function transformOutside(
content: string,
mask: RegExp,
transform: (segment: string) => string,
onMasked: (segment: string) => string = (segment) => segment,
): string {
const pattern = new RegExp(mask.source, mask.flags); // its own lastIndex
let result = '';
let index = 0;
let match: RegExpExecArray | null;
while ((match = pattern.exec(content)) !== null) {
result += transform(content.slice(index, match.index)) + onMasked(match[0]);
index = pattern.lastIndex;
}
return result + transform(content.slice(index));
}
// Runs `transform` over the prose in `content`, leaving code untouched.
//
// The rewrites below are plain regex over the whole document, and both used to
// corrupt the user's own code: `print(row[0])` rendered as
// `print(row[0](#cite-0))`, and `PS1="\[\e[0m\]"` as `PS1="$$\e[0m$$"`. Neither
// can move into a remark plugin: remark-math tokenises `$`/`$$` while parsing,
// so the LaTeX pass has to run on the raw string before remark ever sees it.
export function applyOutsideCode(
content: string,
transform: (segment: string) => string,
): string {
return transformOutside(content, CODE_SPAN, transform);
}
// Rewrites one LaTeX span into the ``$$``/``$`` form remark-math parses.
function toDollarMath(span: string): string {
const equation = span.slice(2, -2);
return span.startsWith('\\[') ? `$$${equation}$$` : `$${equation}$`;
}
// Replaces block-level ``\[ \]`` and inline ``\( \)`` LaTeX delimiters with the
// ``$$``/``$`` forms remark-math understands.
export function preprocessLaTeX(content: string): string {
const blockProcessedContent = content.replace(
/\\\[(.*?)\\\]/gs,
(_, equation) => `$$${equation}$$`,
return applyOutsideCode(content, (prose) =>
prose.replace(LATEX_SPAN, toDollarMath),
);
return blockProcessedContent.replace(
/\\\((.*?)\\\)/gs,
(_, equation) => `$${equation}$`,
}
// Turns citation references ``[N]`` into ``[N](#cite-N)`` links so
// ReactMarkdown renders them as <a> tags we can style. The lookarounds skip
// references that are already links.
function linkCitations(prose: string): string {
return prose.replace(
/(?<!\[)\[(\d+)\](?!\()/g,
(_, num) => `[${num}](#cite-${num})`,
);
}
type ContentSegment = { type: 'text' | 'mermaid'; content: string };
export function processMarkdownContent(content: string): ContentSegment[] {
let processedContent = preprocessLaTeX(content);
// Convert citation references [N] into markdown links [N](#cite-N)
// so ReactMarkdown renders them as <a> tags we can style.
// Avoid matching inside code blocks or existing links.
processedContent = processedContent.replace(
/(?<!\[)\[(\d+)\](?!\()/g,
(_, num) => `[${num}](#cite-${num})`,
// Citations are linked outside code — an index subscript like `row[0]` is not
// a citation — and outside the math this same pass produces. Math the answer
// already wrote as ``$…$`` stays unmasked, since ``$`` is also a currency
// sign. This has to run before the ```mermaid split below, so diagram source
// needs the same guard.
const processedContent = applyOutsideCode(content, (prose) =>
transformOutside(prose, LATEX_SPAN, linkCitations, toDollarMath),
);
const contentSegments: ContentSegment[] = [];