mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-05 12:13:55 +00:00
fix: md citation issue
This commit is contained in:
1 parent
5c55d2610b
commit
2ff895ac9c
2 files changed
+237
-14
No files matched your search
@@ -0,0 +1,166 @@
|
||||
/**
|
||||
* Code must survive the markdown string pre-passes.
|
||||
*
|
||||
* ``processMarkdownContent`` rewrote every ``[N]`` as ``[N](#cite-N)`` and
|
||||
* ``preprocessLaTeX`` every ``\[…\]`` as ``$$…$$``, neither with any awareness
|
||||
* of code, so an index subscript or a bash prompt escape came out corrupted in
|
||||
* the rendered answer — and in what the copy button copied.
|
||||
*
|
||||
* Neither rewrite can move into a remark plugin, where a ``code`` node would be
|
||||
* distinct from a ``text`` one: remark-math tokenises ``$``/``$$`` while
|
||||
* parsing, so the LaTeX pass has to happen on the raw string. Hence the shared
|
||||
* ``applyOutsideCode`` masking helper rather than an AST visitor.
|
||||
*/
|
||||
import { describe, expect, it } from 'vitest';
|
||||
|
||||
import {
|
||||
applyOutsideCode,
|
||||
preprocessLaTeX,
|
||||
processMarkdownContent,
|
||||
} from './MarkdownAnswer';
|
||||
|
||||
/** The rendered text of a single-segment answer. */
|
||||
const text = (source: string) => processMarkdownContent(source)[0].content;
|
||||
|
||||
describe('citation rewriting spares code', () => {
|
||||
it('leaves an index subscript alone inside a fence', () => {
|
||||
const source = '```python\nfor row in cur:\n print(row[0])\n```';
|
||||
expect(text(source)).toBe(source);
|
||||
});
|
||||
|
||||
it('leaves an index subscript alone inside an inline span', () => {
|
||||
expect(text('Use `items[2]` here.')).toBe('Use `items[2]` here.');
|
||||
});
|
||||
|
||||
it('leaves a mid-sentence triple-backtick span alone', () => {
|
||||
expect(text('inline ```code[0]``` end')).toBe('inline ```code[0]``` end');
|
||||
});
|
||||
|
||||
it('leaves an inline span that wraps a line alone', () => {
|
||||
expect(text('use `arr[0]\nnext` here')).toBe('use `arr[0]\nnext` here');
|
||||
});
|
||||
|
||||
it('leaves a tilde fence alone', () => {
|
||||
const source = '~~~python\nprint(row[0])\n~~~';
|
||||
expect(text(source)).toBe(source);
|
||||
});
|
||||
|
||||
it('leaves a fence nested in a longer one alone', () => {
|
||||
// A ``` run does not close a ```` fence, so the inner block is content.
|
||||
const source = '````markdown\n```js\nconst a = arr[0];\n```\n````';
|
||||
expect(text(source)).toBe(source);
|
||||
});
|
||||
|
||||
it('leaves a fence indented under a list item alone', () => {
|
||||
const source =
|
||||
'1. Step:\n\n ```js\n const a = arr[0];\n\n const b = arr[1];\n ```';
|
||||
expect(text(source)).toBe(source);
|
||||
});
|
||||
|
||||
it('leaves an unterminated fence alone while the answer is still streaming', () => {
|
||||
const source = '```python\nprint(row[0]';
|
||||
expect(text(source)).toBe(source);
|
||||
});
|
||||
|
||||
it('leaves an unterminated inline span alone while streaming', () => {
|
||||
expect(text('Use `items[2')).toBe('Use `items[2');
|
||||
});
|
||||
|
||||
it('leaves mermaid node labels alone', () => {
|
||||
// The cite pass runs before the ```mermaid split, so diagram source was
|
||||
// corrupted too and the render failed.
|
||||
const source = '```mermaid\nflowchart TD\n A[0] --> B[1]\n```';
|
||||
expect(processMarkdownContent(source)[0].content).toBe(
|
||||
'flowchart TD\n A[0] --> B[1]',
|
||||
);
|
||||
});
|
||||
|
||||
it('still links a real citation in prose', () => {
|
||||
expect(text('See source [1] for details.')).toBe(
|
||||
'See source [1](#cite-1) for details.',
|
||||
);
|
||||
});
|
||||
|
||||
it('links prose citations on either side of a fence without touching it', () => {
|
||||
expect(
|
||||
text('Per [1], run:\n```js\nconst a = arr[0];\n```\nthen [2].'),
|
||||
).toBe(
|
||||
'Per [1](#cite-1), run:\n```js\nconst a = arr[0];\n```\nthen [2](#cite-2).',
|
||||
);
|
||||
});
|
||||
|
||||
it('still links a citation on an indented list continuation line', () => {
|
||||
// Four-space indents are list content far more often than they are code,
|
||||
// so they are prose to the masker.
|
||||
expect(text('- one\n - nested [1]')).toBe(
|
||||
'- one\n - nested [1](#cite-1)',
|
||||
);
|
||||
});
|
||||
|
||||
it('leaves an already-linked reference alone', () => {
|
||||
expect(text('see [1](#cite-1) ok')).toBe('see [1](#cite-1) ok');
|
||||
});
|
||||
});
|
||||
|
||||
describe('LaTeX preprocessing spares code', () => {
|
||||
it('leaves bash prompt escapes alone', () => {
|
||||
// `\[`/`\]` are non-printing-sequence markers in PS1; the block rule turned
|
||||
// them into `$$` delimiters and broke the prompt.
|
||||
const source = '```bash\nPS1="\\[\\e[0m\\]$ "\n```';
|
||||
expect(preprocessLaTeX(source)).toBe(source);
|
||||
});
|
||||
|
||||
it('leaves bracket character classes alone in an inline span', () => {
|
||||
expect(preprocessLaTeX('match `\\[0-9\\]` here')).toBe(
|
||||
'match `\\[0-9\\]` here',
|
||||
);
|
||||
});
|
||||
|
||||
it('still converts real inline math in prose', () => {
|
||||
expect(preprocessLaTeX('The value \\(x^2\\) is fine.')).toBe(
|
||||
'The value $x^2$ is fine.',
|
||||
);
|
||||
});
|
||||
|
||||
it('still converts real block math in prose', () => {
|
||||
expect(preprocessLaTeX('Given \\[a+b\\] we get')).toBe(
|
||||
'Given $$a+b$$ we get',
|
||||
);
|
||||
});
|
||||
|
||||
it('does not link a subscript inside the math it just produced', () => {
|
||||
// `$a[1](#cite-1)$` is not something KaTeX can render.
|
||||
expect(text('value \\(a[1]\\) end')).toBe('value $a[1]$ end');
|
||||
expect(text('given \\[a[1]+b\\] we get [2]')).toBe(
|
||||
'given $$a[1]+b$$ we get [2](#cite-2)',
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
describe('applyOutsideCode', () => {
|
||||
const shout = (segment: string) => segment.toUpperCase();
|
||||
|
||||
it('transforms prose either side of every code span', () => {
|
||||
expect(applyOutsideCode('a `b` c ```d``` e', shout)).toBe(
|
||||
'A `b` C ```d``` E',
|
||||
);
|
||||
});
|
||||
|
||||
it('handles several inline spans on one line', () => {
|
||||
expect(applyOutsideCode('x `a` y `b` z', shout)).toBe('X `a` Y `b` Z');
|
||||
});
|
||||
|
||||
it('treats a double-backtick span containing a backtick as one span', () => {
|
||||
expect(applyOutsideCode('x ``a`b`` y', shout)).toBe('X ``a`b`` Y');
|
||||
});
|
||||
|
||||
it('does not close a long fence on a shorter run', () => {
|
||||
expect(applyOutsideCode('````\na\n```\nb\n````\nc', shout)).toBe(
|
||||
'````\na\n```\nb\n````\nC',
|
||||
);
|
||||
});
|
||||
|
||||
it('passes an empty string through', () => {
|
||||
expect(applyOutsideCode('', shout)).toBe('');
|
||||
});
|
||||
});
|
||||
@@ -22,30 +22,87 @@ import {
|
||||
sandboxUrlTransform,
|
||||
} from './sandboxLinks';
|
||||
|
||||
// One fenced block or inline code span. Backtick runs are length-matched, so a
|
||||
// ```` fence closes only on ```` and nested fences stay masked. The
|
||||
// unterminated alternatives keep an open span matched too: answers stream in,
|
||||
// so a fence is open for most of its life and needs protecting the whole time,
|
||||
// not just once the closing fence arrives. Four-space indented blocks are
|
||||
// deliberately not masked — list continuation lines are indented the same way,
|
||||
// and masking those would drop real citations out of nested lists.
|
||||
const CODE_SPAN =
|
||||
/(?<![^\n])[ \t]*(`{3,})[^`\n]*(?:[\s\S]*?\n[ \t]*\1`*[ \t]*(?![^\n])|[\s\S]*$)|(?<![^\n])[ \t]*(~{3,})[^\n]*(?:[\s\S]*?\n[ \t]*\2~*[ \t]*(?![^\n])|[\s\S]*$)|(`+)(?:[^\n]|\n(?![ \t]*\n))*?\3(?!`)|`+[^`\n]*$/g;
|
||||
|
||||
// ``\[ \]`` and ``\( \)`` LaTeX, which remark-math does not recognise.
|
||||
const LATEX_SPAN = /\\\[[\s\S]*?\\\]|\\\([\s\S]*?\\\)/g;
|
||||
|
||||
// Applies `transform` to everything outside the regions `mask` matches, and
|
||||
// `onMasked` to those regions themselves.
|
||||
function transformOutside(
|
||||
content: string,
|
||||
mask: RegExp,
|
||||
transform: (segment: string) => string,
|
||||
onMasked: (segment: string) => string = (segment) => segment,
|
||||
): string {
|
||||
const pattern = new RegExp(mask.source, mask.flags); // its own lastIndex
|
||||
let result = '';
|
||||
let index = 0;
|
||||
let match: RegExpExecArray | null;
|
||||
|
||||
while ((match = pattern.exec(content)) !== null) {
|
||||
result += transform(content.slice(index, match.index)) + onMasked(match[0]);
|
||||
index = pattern.lastIndex;
|
||||
}
|
||||
return result + transform(content.slice(index));
|
||||
}
|
||||
|
||||
// Runs `transform` over the prose in `content`, leaving code untouched.
|
||||
//
|
||||
// The rewrites below are plain regex over the whole document, and both used to
|
||||
// corrupt the user's own code: `print(row[0])` rendered as
|
||||
// `print(row[0](#cite-0))`, and `PS1="\[\e[0m\]"` as `PS1="$$\e[0m$$"`. Neither
|
||||
// can move into a remark plugin: remark-math tokenises `$`/`$$` while parsing,
|
||||
// so the LaTeX pass has to run on the raw string before remark ever sees it.
|
||||
export function applyOutsideCode(
|
||||
content: string,
|
||||
transform: (segment: string) => string,
|
||||
): string {
|
||||
return transformOutside(content, CODE_SPAN, transform);
|
||||
}
|
||||
|
||||
// Rewrites one LaTeX span into the ``$$``/``$`` form remark-math parses.
|
||||
function toDollarMath(span: string): string {
|
||||
const equation = span.slice(2, -2);
|
||||
return span.startsWith('\\[') ? `$$${equation}$$` : `$${equation}$`;
|
||||
}
|
||||
|
||||
// Replaces block-level ``\[ \]`` and inline ``\( \)`` LaTeX delimiters with the
|
||||
// ``$$``/``$`` forms remark-math understands.
|
||||
export function preprocessLaTeX(content: string): string {
|
||||
const blockProcessedContent = content.replace(
|
||||
/\\\[(.*?)\\\]/gs,
|
||||
(_, equation) => `$$${equation}$$`,
|
||||
return applyOutsideCode(content, (prose) =>
|
||||
prose.replace(LATEX_SPAN, toDollarMath),
|
||||
);
|
||||
return blockProcessedContent.replace(
|
||||
/\\\((.*?)\\\)/gs,
|
||||
(_, equation) => `$${equation}$`,
|
||||
}
|
||||
|
||||
// Turns citation references ``[N]`` into ``[N](#cite-N)`` links so
|
||||
// ReactMarkdown renders them as <a> tags we can style. The lookarounds skip
|
||||
// references that are already links.
|
||||
function linkCitations(prose: string): string {
|
||||
return prose.replace(
|
||||
/(?<!\[)\[(\d+)\](?!\()/g,
|
||||
(_, num) => `[${num}](#cite-${num})`,
|
||||
);
|
||||
}
|
||||
|
||||
type ContentSegment = { type: 'text' | 'mermaid'; content: string };
|
||||
|
||||
export function processMarkdownContent(content: string): ContentSegment[] {
|
||||
let processedContent = preprocessLaTeX(content);
|
||||
|
||||
// Convert citation references [N] into markdown links [N](#cite-N)
|
||||
// so ReactMarkdown renders them as <a> tags we can style.
|
||||
// Avoid matching inside code blocks or existing links.
|
||||
processedContent = processedContent.replace(
|
||||
/(?<!\[)\[(\d+)\](?!\()/g,
|
||||
(_, num) => `[${num}](#cite-${num})`,
|
||||
// Citations are linked outside code — an index subscript like `row[0]` is not
|
||||
// a citation — and outside the math this same pass produces. Math the answer
|
||||
// already wrote as ``$…$`` stays unmasked, since ``$`` is also a currency
|
||||
// sign. This has to run before the ```mermaid split below, so diagram source
|
||||
// needs the same guard.
|
||||
const processedContent = applyOutsideCode(content, (prose) =>
|
||||
transformOutside(prose, LATEX_SPAN, linkCitations, toDollarMath),
|
||||
);
|
||||
|
||||
const contentSegments: ContentSegment[] = [];
|
||||
|
||||
Reference in new issue
Block a user