feat(webui): support latex \(...\) and \[...\] math delimiters in markdown
This commit is contained in:
@@ -15,13 +15,17 @@ import {
|
||||
dispatchFileLink,
|
||||
FILE_LINK_HREF_PREFIX,
|
||||
rehypePathLinks,
|
||||
resolveChatImageSrc,
|
||||
} from "@/lib/fileLink";
|
||||
import remarkLatexMathDelimiters from "@/lib/remarkLatexMathDelimiters";
|
||||
|
||||
interface MarkdownContentProps {
|
||||
content: string;
|
||||
className?: string;
|
||||
/** When true, defer expensive renders (e.g. mermaid) until streaming ends. */
|
||||
isStreaming?: boolean;
|
||||
/** Current conversation thread; enables serving workspace images inline. */
|
||||
threadId?: string | null;
|
||||
}
|
||||
|
||||
const sanitizeSchema = {
|
||||
@@ -36,7 +40,7 @@ const sanitizeSchema = {
|
||||
};
|
||||
|
||||
export const MarkdownContent = React.memo<MarkdownContentProps>(
|
||||
({ content, className = "", isStreaming = false }) => {
|
||||
({ content, className = "", isStreaming = false, threadId = null }) => {
|
||||
return (
|
||||
<div
|
||||
className={cn(
|
||||
@@ -45,7 +49,7 @@ export const MarkdownContent = React.memo<MarkdownContentProps>(
|
||||
)}
|
||||
>
|
||||
<ReactMarkdown
|
||||
remarkPlugins={[remarkGfm, remarkMath]}
|
||||
remarkPlugins={[remarkGfm, remarkMath, remarkLatexMathDelimiters]}
|
||||
rehypePlugins={[
|
||||
rehypeRaw,
|
||||
rehypePathLinks,
|
||||
@@ -152,6 +156,24 @@ export const MarkdownContent = React.memo<MarkdownContentProps>(
|
||||
</button>
|
||||
);
|
||||
}
|
||||
// Agent-written markdown links that reference workspace files
|
||||
// (`[x](artifacts/a.png)`, or a host-absolute path under the
|
||||
// conversation files dir) must open the file dialog — passing
|
||||
// them through lets the browser resolve them against the page
|
||||
// origin, which 404s.
|
||||
const fileLink = href ? detectFileLink(href) : null;
|
||||
if (fileLink) {
|
||||
return (
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => dispatchFileLink(fileLink)}
|
||||
title={`Open ${fileLink.display}`}
|
||||
className="text-primary underline underline-offset-2 hover:text-foreground focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-ring"
|
||||
>
|
||||
{children}
|
||||
</button>
|
||||
);
|
||||
}
|
||||
return (
|
||||
<a
|
||||
href={href}
|
||||
@@ -163,6 +185,22 @@ export const MarkdownContent = React.memo<MarkdownContentProps>(
|
||||
</a>
|
||||
);
|
||||
},
|
||||
img({ src, alt }: React.ImgHTMLAttributes<HTMLImageElement>) {
|
||||
// Workspace-relative image paths (artifacts/x.png) would resolve
|
||||
// against the page origin and 404 — serve them through the
|
||||
// conversation's workspace file endpoint instead.
|
||||
const raw = typeof src === "string" ? src : undefined;
|
||||
const resolved = resolveChatImageSrc(raw, threadId);
|
||||
if (!resolved) return null;
|
||||
return (
|
||||
// eslint-disable-next-line @next/next/no-img-element
|
||||
<img
|
||||
src={resolved}
|
||||
alt={alt ?? ""}
|
||||
className="my-4 max-w-full rounded-md border border-border"
|
||||
/>
|
||||
);
|
||||
},
|
||||
blockquote({ children }: { children?: React.ReactNode }) {
|
||||
return (
|
||||
<blockquote className="my-4 border-l-4 border-border pl-4 italic text-[var(--color-text-tertiary)]">
|
||||
|
||||
@@ -0,0 +1,166 @@
|
||||
import { unified } from "unified";
|
||||
import remarkParse from "remark-parse";
|
||||
import remarkMath from "remark-math";
|
||||
import { describe, expect, it } from "vitest";
|
||||
|
||||
import remarkLatexMathDelimiters, {
|
||||
normalizeLatexMathDelimiters,
|
||||
} from "./remarkLatexMathDelimiters";
|
||||
|
||||
interface Node {
|
||||
type: string;
|
||||
value?: string;
|
||||
children?: Node[];
|
||||
}
|
||||
|
||||
function collect(tree: Node, types: Set<string>): Node[] {
|
||||
const found: Node[] = [];
|
||||
const walk = (n: Node) => {
|
||||
if (types.has(n.type)) found.push(n);
|
||||
n.children?.forEach(walk);
|
||||
};
|
||||
walk(tree);
|
||||
return found;
|
||||
}
|
||||
|
||||
function parse(src: string): Node {
|
||||
const p = unified()
|
||||
.use(remarkParse)
|
||||
.use(remarkMath)
|
||||
.use(remarkLatexMathDelimiters);
|
||||
return p.runSync(p.parse(src)) as unknown as Node;
|
||||
}
|
||||
|
||||
describe("normalizeLatexMathDelimiters", () => {
|
||||
it("rewrites inline math", () => {
|
||||
expect(normalizeLatexMathDelimiters("a \\(\\theta_0\\) b")).toBe(
|
||||
"a $\\theta_0$ b"
|
||||
);
|
||||
});
|
||||
|
||||
it("trims edge spaces inside inline math (pandoc $ rules)", () => {
|
||||
expect(normalizeLatexMathDelimiters("\\( x \\)")).toBe("$x$");
|
||||
});
|
||||
|
||||
it("rewrites display math to a $$ flow block", () => {
|
||||
expect(normalizeLatexMathDelimiters("\\[ x \\]")).toBe("\n\n$$\nx\n$$\n\n");
|
||||
});
|
||||
|
||||
it("keeps multiline display bodies", () => {
|
||||
const out = normalizeLatexMathDelimiters("\\[\na+b\n\\]");
|
||||
expect(out).toBe("\n\n$$\na+b\n$$\n\n");
|
||||
});
|
||||
|
||||
it("leaves fenced code untouched", () => {
|
||||
const src = "```tex\n\\(x\\) and \\[y\\]\n```";
|
||||
expect(normalizeLatexMathDelimiters(src)).toBe(src);
|
||||
});
|
||||
|
||||
it("leaves tilde fences and longer fences untouched", () => {
|
||||
const src = "~~~~\n\\(x\\)\n~~~~\n\nafter \\(y\\)";
|
||||
expect(normalizeLatexMathDelimiters(src)).toBe(
|
||||
"~~~~\n\\(x\\)\n~~~~\n\nafter $y$"
|
||||
);
|
||||
});
|
||||
|
||||
it("leaves inline code spans untouched", () => {
|
||||
const src = "`\\(x\\)` and ``\\[y\\]`` but \\(z\\)";
|
||||
expect(normalizeLatexMathDelimiters(src)).toBe(
|
||||
"`\\(x\\)` and ``\\[y\\]`` but $z$"
|
||||
);
|
||||
});
|
||||
|
||||
it("leaves unclosed delimiters as literal text", () => {
|
||||
expect(normalizeLatexMathDelimiters("\\(never closed")).toBe(
|
||||
"\\(never closed"
|
||||
);
|
||||
});
|
||||
|
||||
it("handles unclosed fences by copying to EOF", () => {
|
||||
const src = "```\n\\(x\\)";
|
||||
expect(normalizeLatexMathDelimiters(src)).toBe(src);
|
||||
});
|
||||
|
||||
it("normalizes the reported radar-error example", () => {
|
||||
const src =
|
||||
"偏转了 \\(\\theta_0\\),则:\n\n\\[ \\hat{\\theta}\\approx\\theta-\\theta_0 \\]";
|
||||
const out = normalizeLatexMathDelimiters(src);
|
||||
expect(out).toContain("$\\theta_0$");
|
||||
expect(out).toContain("$$\n\\hat{\\theta}\\approx\\theta-\\theta_0\n$$");
|
||||
});
|
||||
});
|
||||
|
||||
describe("remarkLatexMathDelimiters (parse integration)", () => {
|
||||
it("parses \\(...\\) as inlineMath", () => {
|
||||
const math = collect(parse("a \\(\\theta_0\\) b"), new Set(["inlineMath"]));
|
||||
expect(math).toHaveLength(1);
|
||||
expect(math[0].value).toBe("\\theta_0");
|
||||
});
|
||||
|
||||
it("parses \\[...\\] as display math", () => {
|
||||
const math = collect(
|
||||
parse("\\[ \\hat{\\theta}\\approx\\theta \\]"),
|
||||
new Set(["math"])
|
||||
);
|
||||
expect(math).toHaveLength(1);
|
||||
expect(math[0].value).toBe("\\hat{\\theta}\\approx\\theta");
|
||||
});
|
||||
|
||||
it("does not convert inside inline code", () => {
|
||||
const tree = parse("`\\(x\\)`");
|
||||
expect(collect(tree, new Set(["inlineMath"]))).toHaveLength(0);
|
||||
expect(collect(tree, new Set(["inlineCode"]))[0].value).toBe("\\(x\\)");
|
||||
});
|
||||
|
||||
it("does not convert inside fenced code", () => {
|
||||
const tree = parse("```\n\\(x\\)\n```");
|
||||
expect(collect(tree, new Set(["inlineMath", "math"]))).toHaveLength(0);
|
||||
});
|
||||
|
||||
it("still parses native $ and $$ forms", () => {
|
||||
const tree = parse("$a$ and\n\n$$\nb\n$$");
|
||||
expect(collect(tree, new Set(["inlineMath"]))).toHaveLength(1);
|
||||
expect(collect(tree, new Set(["math"]))).toHaveLength(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("KaTeX end-to-end (reported chat content)", () => {
|
||||
it("renders inline and display math to katex elements", async () => {
|
||||
const { default: remarkGfm } = await import("remark-gfm");
|
||||
const { default: remarkRehype } = await import("remark-rehype");
|
||||
const { default: rehypeKatex } = await import("rehype-katex");
|
||||
|
||||
const src = [
|
||||
"例如整个雷达阵列机械安装偏转了 \\(\\theta_0\\),则:",
|
||||
"",
|
||||
"\\[ \\hat{\\theta}\\approx\\theta-\\theta_0 \\]",
|
||||
"",
|
||||
"\\[ H_n^{\\mathrm{meas}} = g_n H_n^{\\mathrm{ideal}} \\]",
|
||||
].join("\n");
|
||||
|
||||
const processor = unified()
|
||||
.use(remarkParse)
|
||||
.use(remarkGfm)
|
||||
.use(remarkMath)
|
||||
.use(remarkLatexMathDelimiters)
|
||||
.use(remarkRehype)
|
||||
.use(rehypeKatex);
|
||||
const hast = processor.runSync(processor.parse(src)) as unknown as Node & {
|
||||
properties?: { className?: string[] };
|
||||
};
|
||||
|
||||
const classes: string[] = [];
|
||||
const walk = (n: Node & { properties?: { className?: string[] } }) => {
|
||||
const c = n.properties?.className;
|
||||
if (Array.isArray(c)) classes.push(...c);
|
||||
n.children?.forEach((child) =>
|
||||
walk(child as Node & { properties?: { className?: string[] } })
|
||||
);
|
||||
};
|
||||
walk(hast);
|
||||
|
||||
expect(classes).toContain("katex");
|
||||
expect(classes).toContain("katex-display");
|
||||
expect(classes.filter((c) => c === "katex-display")).toHaveLength(2);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,100 @@
|
||||
// remarkLatexMathDelimiters: accept LaTeX `\(...\)` / `\[...\]` math
|
||||
// delimiters in chat markdown.
|
||||
//
|
||||
// Chat models frequently emit LaTeX's native delimiters instead of the
|
||||
// `$...$` / `$$...$$` forms remark-math understands. Two problems result:
|
||||
// the math never reaches KaTeX, and CommonMark then eats the backslash
|
||||
// (`\(` is a valid escape of punctuation), so the reader sees a mangled
|
||||
// `(\theta_0)`.
|
||||
//
|
||||
// A normal remark transformer runs too late — by mdast time the backslashes
|
||||
// are already gone — so this plugin wraps `this.parser` and normalizes the
|
||||
// raw source before parsing: `\(...\)` → `$...$`, `\[...\]` → a `$$` flow
|
||||
// block. The scanner copies fenced code blocks and inline code spans
|
||||
// verbatim so literal `\(` in code is preserved. remark-math must also be
|
||||
// in the pipeline to parse the resulting `$`/`$$` spans.
|
||||
|
||||
import type { Root } from "mdast";
|
||||
import type { Plugin } from "unified";
|
||||
|
||||
const OPENERS: Record<string, string> = { "(": ")", "[": "]" };
|
||||
|
||||
/**
|
||||
* Rewrite LaTeX math delimiters in raw markdown to remark-math's `$` forms.
|
||||
* Pure function, exported for testing.
|
||||
*/
|
||||
export function normalizeLatexMathDelimiters(src: string): string {
|
||||
let out = "";
|
||||
let i = 0;
|
||||
const n = src.length;
|
||||
|
||||
const atLineStart = (pos: number) => pos === 0 || src[pos - 1] === "\n";
|
||||
|
||||
while (i < n) {
|
||||
// Fenced code block: up to 3 spaces of indent, then a run of >=3
|
||||
// backticks or tildes; copy through the closing fence (or EOF).
|
||||
if (atLineStart(i)) {
|
||||
const fence = /^ {0,3}(`{3,}|~{3,})/.exec(src.slice(i));
|
||||
if (fence) {
|
||||
const marker = fence[1];
|
||||
const closeRe = new RegExp(
|
||||
`\\n {0,3}${marker[0] === "`" ? "`" : "~"}{${marker.length},}[^\\S\\n]*(?=\\n|$)`,
|
||||
"g"
|
||||
);
|
||||
closeRe.lastIndex = i;
|
||||
const close = closeRe.exec(src);
|
||||
const end = close ? closeRe.lastIndex : n;
|
||||
out += src.slice(i, end);
|
||||
i = end;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
// Inline code span: a run of N backticks closes on the next run of
|
||||
// exactly N backticks.
|
||||
if (src[i] === "`") {
|
||||
let runEnd = i;
|
||||
while (runEnd < n && src[runEnd] === "`") runEnd++;
|
||||
const ticks = src.slice(i, runEnd);
|
||||
const close = src.indexOf(ticks, runEnd);
|
||||
const end = close === -1 ? runEnd : close + ticks.length;
|
||||
out += src.slice(i, end);
|
||||
i = end;
|
||||
continue;
|
||||
}
|
||||
|
||||
// Math opener: `\(` or `\[`. The first matching closer ends the span —
|
||||
// LaTeX bodies don't legitimately contain `\)` / `\]`.
|
||||
if (src[i] === "\\" && i + 1 < n && src[i + 1] in OPENERS) {
|
||||
const open = src[i + 1];
|
||||
const closer = `\\${OPENERS[open]}`;
|
||||
const close = src.indexOf(closer, i + 2);
|
||||
if (close !== -1) {
|
||||
const inner = src.slice(i + 2, close).trim();
|
||||
if (open === "(") {
|
||||
// Single-`$` inline math (pandoc rules) rejects inner edge spaces.
|
||||
out += `$${inner}$`;
|
||||
} else {
|
||||
// Display math: flow `$$` block. Blank lines detach it from any
|
||||
// surrounding paragraph, which matches `\[` semantics.
|
||||
out += `\n\n$$\n${inner}\n$$\n\n`;
|
||||
}
|
||||
i = close + 2;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
out += src[i];
|
||||
i++;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
const remarkLatexMathDelimiters: Plugin<[], Root> = function () {
|
||||
const parser = this.parser;
|
||||
if (!parser) return;
|
||||
this.parser = (doc, file) =>
|
||||
parser(normalizeLatexMathDelimiters(String(doc)), file);
|
||||
};
|
||||
|
||||
export default remarkLatexMathDelimiters;
|
||||
Reference in New Issue
Block a user