From 84e4d298b3710dedb721926bfc0dfccde7f9fa36 Mon Sep 17 00:00:00 2001 From: m4 Date: Tue, 11 Aug 2026 15:07:53 +0800 Subject: [PATCH] feat(webui): support latex \(...\) and \[...\] math delimiters in markdown --- src/app/components/MarkdownContent.tsx | 42 +++++- src/lib/remarkLatexMathDelimiters.test.ts | 166 ++++++++++++++++++++++ src/lib/remarkLatexMathDelimiters.ts | 100 +++++++++++++ 3 files changed, 306 insertions(+), 2 deletions(-) create mode 100644 src/lib/remarkLatexMathDelimiters.test.ts create mode 100644 src/lib/remarkLatexMathDelimiters.ts diff --git a/src/app/components/MarkdownContent.tsx b/src/app/components/MarkdownContent.tsx index 8791022..6f30fbe 100644 --- a/src/app/components/MarkdownContent.tsx +++ b/src/app/components/MarkdownContent.tsx @@ -15,13 +15,17 @@ import { dispatchFileLink, FILE_LINK_HREF_PREFIX, rehypePathLinks, + resolveChatImageSrc, } from "@/lib/fileLink"; +import remarkLatexMathDelimiters from "@/lib/remarkLatexMathDelimiters"; interface MarkdownContentProps { content: string; className?: string; /** When true, defer expensive renders (e.g. mermaid) until streaming ends. */ isStreaming?: boolean; + /** Current conversation thread; enables serving workspace images inline. */ + threadId?: string | null; } const sanitizeSchema = { @@ -36,7 +40,7 @@ const sanitizeSchema = { }; export const MarkdownContent = React.memo( - ({ content, className = "", isStreaming = false }) => { + ({ content, className = "", isStreaming = false, threadId = null }) => { return (
( )} > ( ); } + // Agent-written markdown links that reference workspace files + // (`[x](artifacts/a.png)`, or a host-absolute path under the + // conversation files dir) must open the file dialog — passing + // them through lets the browser resolve them against the page + // origin, which 404s. + const fileLink = href ? detectFileLink(href) : null; + if (fileLink) { + return ( + + ); + } return ( ( ); }, + img({ src, alt }: React.ImgHTMLAttributes) { + // Workspace-relative image paths (artifacts/x.png) would resolve + // against the page origin and 404 — serve them through the + // conversation's workspace file endpoint instead. + const raw = typeof src === "string" ? src : undefined; + const resolved = resolveChatImageSrc(raw, threadId); + if (!resolved) return null; + return ( + // eslint-disable-next-line @next/next/no-img-element + {alt + ); + }, blockquote({ children }: { children?: React.ReactNode }) { return (
diff --git a/src/lib/remarkLatexMathDelimiters.test.ts b/src/lib/remarkLatexMathDelimiters.test.ts new file mode 100644 index 0000000..35ccc9b --- /dev/null +++ b/src/lib/remarkLatexMathDelimiters.test.ts @@ -0,0 +1,166 @@ +import { unified } from "unified"; +import remarkParse from "remark-parse"; +import remarkMath from "remark-math"; +import { describe, expect, it } from "vitest"; + +import remarkLatexMathDelimiters, { + normalizeLatexMathDelimiters, +} from "./remarkLatexMathDelimiters"; + +interface Node { + type: string; + value?: string; + children?: Node[]; +} + +function collect(tree: Node, types: Set): Node[] { + const found: Node[] = []; + const walk = (n: Node) => { + if (types.has(n.type)) found.push(n); + n.children?.forEach(walk); + }; + walk(tree); + return found; +} + +function parse(src: string): Node { + const p = unified() + .use(remarkParse) + .use(remarkMath) + .use(remarkLatexMathDelimiters); + return p.runSync(p.parse(src)) as unknown as Node; +} + +describe("normalizeLatexMathDelimiters", () => { + it("rewrites inline math", () => { + expect(normalizeLatexMathDelimiters("a \\(\\theta_0\\) b")).toBe( + "a $\\theta_0$ b" + ); + }); + + it("trims edge spaces inside inline math (pandoc $ rules)", () => { + expect(normalizeLatexMathDelimiters("\\( x \\)")).toBe("$x$"); + }); + + it("rewrites display math to a $$ flow block", () => { + expect(normalizeLatexMathDelimiters("\\[ x \\]")).toBe("\n\n$$\nx\n$$\n\n"); + }); + + it("keeps multiline display bodies", () => { + const out = normalizeLatexMathDelimiters("\\[\na+b\n\\]"); + expect(out).toBe("\n\n$$\na+b\n$$\n\n"); + }); + + it("leaves fenced code untouched", () => { + const src = "```tex\n\\(x\\) and \\[y\\]\n```"; + expect(normalizeLatexMathDelimiters(src)).toBe(src); + }); + + it("leaves tilde fences and longer fences untouched", () => { + const src = "~~~~\n\\(x\\)\n~~~~\n\nafter \\(y\\)"; + expect(normalizeLatexMathDelimiters(src)).toBe( + "~~~~\n\\(x\\)\n~~~~\n\nafter $y$" + ); + }); + + it("leaves inline code spans untouched", () => { + const src = "`\\(x\\)` and ``\\[y\\]`` but \\(z\\)"; + expect(normalizeLatexMathDelimiters(src)).toBe( + "`\\(x\\)` and ``\\[y\\]`` but $z$" + ); + }); + + it("leaves unclosed delimiters as literal text", () => { + expect(normalizeLatexMathDelimiters("\\(never closed")).toBe( + "\\(never closed" + ); + }); + + it("handles unclosed fences by copying to EOF", () => { + const src = "```\n\\(x\\)"; + expect(normalizeLatexMathDelimiters(src)).toBe(src); + }); + + it("normalizes the reported radar-error example", () => { + const src = + "偏转了 \\(\\theta_0\\),则:\n\n\\[ \\hat{\\theta}\\approx\\theta-\\theta_0 \\]"; + const out = normalizeLatexMathDelimiters(src); + expect(out).toContain("$\\theta_0$"); + expect(out).toContain("$$\n\\hat{\\theta}\\approx\\theta-\\theta_0\n$$"); + }); +}); + +describe("remarkLatexMathDelimiters (parse integration)", () => { + it("parses \\(...\\) as inlineMath", () => { + const math = collect(parse("a \\(\\theta_0\\) b"), new Set(["inlineMath"])); + expect(math).toHaveLength(1); + expect(math[0].value).toBe("\\theta_0"); + }); + + it("parses \\[...\\] as display math", () => { + const math = collect( + parse("\\[ \\hat{\\theta}\\approx\\theta \\]"), + new Set(["math"]) + ); + expect(math).toHaveLength(1); + expect(math[0].value).toBe("\\hat{\\theta}\\approx\\theta"); + }); + + it("does not convert inside inline code", () => { + const tree = parse("`\\(x\\)`"); + expect(collect(tree, new Set(["inlineMath"]))).toHaveLength(0); + expect(collect(tree, new Set(["inlineCode"]))[0].value).toBe("\\(x\\)"); + }); + + it("does not convert inside fenced code", () => { + const tree = parse("```\n\\(x\\)\n```"); + expect(collect(tree, new Set(["inlineMath", "math"]))).toHaveLength(0); + }); + + it("still parses native $ and $$ forms", () => { + const tree = parse("$a$ and\n\n$$\nb\n$$"); + expect(collect(tree, new Set(["inlineMath"]))).toHaveLength(1); + expect(collect(tree, new Set(["math"]))).toHaveLength(1); + }); +}); + +describe("KaTeX end-to-end (reported chat content)", () => { + it("renders inline and display math to katex elements", async () => { + const { default: remarkGfm } = await import("remark-gfm"); + const { default: remarkRehype } = await import("remark-rehype"); + const { default: rehypeKatex } = await import("rehype-katex"); + + const src = [ + "例如整个雷达阵列机械安装偏转了 \\(\\theta_0\\),则:", + "", + "\\[ \\hat{\\theta}\\approx\\theta-\\theta_0 \\]", + "", + "\\[ H_n^{\\mathrm{meas}} = g_n H_n^{\\mathrm{ideal}} \\]", + ].join("\n"); + + const processor = unified() + .use(remarkParse) + .use(remarkGfm) + .use(remarkMath) + .use(remarkLatexMathDelimiters) + .use(remarkRehype) + .use(rehypeKatex); + const hast = processor.runSync(processor.parse(src)) as unknown as Node & { + properties?: { className?: string[] }; + }; + + const classes: string[] = []; + const walk = (n: Node & { properties?: { className?: string[] } }) => { + const c = n.properties?.className; + if (Array.isArray(c)) classes.push(...c); + n.children?.forEach((child) => + walk(child as Node & { properties?: { className?: string[] } }) + ); + }; + walk(hast); + + expect(classes).toContain("katex"); + expect(classes).toContain("katex-display"); + expect(classes.filter((c) => c === "katex-display")).toHaveLength(2); + }); +}); diff --git a/src/lib/remarkLatexMathDelimiters.ts b/src/lib/remarkLatexMathDelimiters.ts new file mode 100644 index 0000000..80c6623 --- /dev/null +++ b/src/lib/remarkLatexMathDelimiters.ts @@ -0,0 +1,100 @@ +// remarkLatexMathDelimiters: accept LaTeX `\(...\)` / `\[...\]` math +// delimiters in chat markdown. +// +// Chat models frequently emit LaTeX's native delimiters instead of the +// `$...$` / `$$...$$` forms remark-math understands. Two problems result: +// the math never reaches KaTeX, and CommonMark then eats the backslash +// (`\(` is a valid escape of punctuation), so the reader sees a mangled +// `(\theta_0)`. +// +// A normal remark transformer runs too late — by mdast time the backslashes +// are already gone — so this plugin wraps `this.parser` and normalizes the +// raw source before parsing: `\(...\)` → `$...$`, `\[...\]` → a `$$` flow +// block. The scanner copies fenced code blocks and inline code spans +// verbatim so literal `\(` in code is preserved. remark-math must also be +// in the pipeline to parse the resulting `$`/`$$` spans. + +import type { Root } from "mdast"; +import type { Plugin } from "unified"; + +const OPENERS: Record = { "(": ")", "[": "]" }; + +/** + * Rewrite LaTeX math delimiters in raw markdown to remark-math's `$` forms. + * Pure function, exported for testing. + */ +export function normalizeLatexMathDelimiters(src: string): string { + let out = ""; + let i = 0; + const n = src.length; + + const atLineStart = (pos: number) => pos === 0 || src[pos - 1] === "\n"; + + while (i < n) { + // Fenced code block: up to 3 spaces of indent, then a run of >=3 + // backticks or tildes; copy through the closing fence (or EOF). + if (atLineStart(i)) { + const fence = /^ {0,3}(`{3,}|~{3,})/.exec(src.slice(i)); + if (fence) { + const marker = fence[1]; + const closeRe = new RegExp( + `\\n {0,3}${marker[0] === "`" ? "`" : "~"}{${marker.length},}[^\\S\\n]*(?=\\n|$)`, + "g" + ); + closeRe.lastIndex = i; + const close = closeRe.exec(src); + const end = close ? closeRe.lastIndex : n; + out += src.slice(i, end); + i = end; + continue; + } + } + + // Inline code span: a run of N backticks closes on the next run of + // exactly N backticks. + if (src[i] === "`") { + let runEnd = i; + while (runEnd < n && src[runEnd] === "`") runEnd++; + const ticks = src.slice(i, runEnd); + const close = src.indexOf(ticks, runEnd); + const end = close === -1 ? runEnd : close + ticks.length; + out += src.slice(i, end); + i = end; + continue; + } + + // Math opener: `\(` or `\[`. The first matching closer ends the span — + // LaTeX bodies don't legitimately contain `\)` / `\]`. + if (src[i] === "\\" && i + 1 < n && src[i + 1] in OPENERS) { + const open = src[i + 1]; + const closer = `\\${OPENERS[open]}`; + const close = src.indexOf(closer, i + 2); + if (close !== -1) { + const inner = src.slice(i + 2, close).trim(); + if (open === "(") { + // Single-`$` inline math (pandoc rules) rejects inner edge spaces. + out += `$${inner}$`; + } else { + // Display math: flow `$$` block. Blank lines detach it from any + // surrounding paragraph, which matches `\[` semantics. + out += `\n\n$$\n${inner}\n$$\n\n`; + } + i = close + 2; + continue; + } + } + + out += src[i]; + i++; + } + return out; +} + +const remarkLatexMathDelimiters: Plugin<[], Root> = function () { + const parser = this.parser; + if (!parser) return; + this.parser = (doc, file) => + parser(normalizeLatexMathDelimiters(String(doc)), file); +}; + +export default remarkLatexMathDelimiters;