fix(desktop): retain Markdown boundary and code-copy whitespace

This commit is contained in:
Teknium
2026-09-07 02:13:09 -07:00
parent 5db3d4b6bd
commit 55ad5afa12
11 changed files with 50 additions and 55 deletions
@@ -123,19 +123,20 @@ export const SyntaxHighlighter: FC<HermesSyntaxHighlighterProps> = ({
defer = false
}) => {
const { t } = useI18n()
const trimmed = (code ?? '').replace(/^\n+/, '').trimEnd()
// Preserve the parser payload for both display and copy, including whitespace.
const content = code ?? ''
// Streaming may hand us empty/incomplete fences — render nothing rather
// than a transient empty card.
if (!trimmed.trim()) {
if (!content.trim()) {
return null
}
if (isLikelyProseCodeBlock(language, trimmed)) {
return <div className="aui-prose-fence whitespace-pre-wrap wrap-anywhere text-foreground">{trimmed}</div>
if (isLikelyProseCodeBlock(language, content)) {
return <div className="aui-prose-fence whitespace-pre-wrap wrap-anywhere text-foreground">{content}</div>
}
const plain = defer || exceedsHighlightBudget(trimmed)
const plain = defer || exceedsHighlightBudget(content)
return (
<CodeCard data-streaming={defer ? 'true' : undefined}>
@@ -145,15 +146,15 @@ export const SyntaxHighlighter: FC<HermesSyntaxHighlighterProps> = ({
iconClassName="size-2.5"
label={t.assistant.tool.copyCode}
showLabel={false}
text={trimmed}
text={content}
/>
<CodeCardBody className="[&_pre]:px-3 [&_pre]:py-2.5">
<ExpandableBlock>
<Pre className="aui-shiki m-0 overflow-hidden bg-transparent p-0">
{plain ? (
<PlainCode code={trimmed} />
<PlainCode code={content} />
) : (
<LazyShiki code={trimmed} colorReplacements={SHIKI_COLOR_REPLACEMENTS} language={language || 'text'} />
<LazyShiki code={content} colorReplacements={SHIKI_COLOR_REPLACEMENTS} language={language || 'text'} />
)}
</Pre>
</ExpandableBlock>
+2 -2
View File
@@ -224,7 +224,7 @@ describe('toChatMessages', () => {
expect(toolPart?.result).toMatchObject({ image: 'https://cdn.example/cat.png', success: true })
// The duplicated image is stripped, but the agent's words survive.
expect(chatMessageText(message)).toBe('Here you go.')
expect(chatMessageText(message)).toBe('Here you go.\n\n')
})
it('lifts @image directive lines into attachmentRefs instead of inline text', () => {
@@ -270,7 +270,7 @@ describe('toChatMessages', () => {
}
])
expect(chatMessageText(message)).toBe('what is in this photo?')
expect(chatMessageText(message)).toBe('what is in this photo?\n')
expect((message as { attachmentRefs?: string[] }).attachmentRefs).toEqual([ref])
})
+2 -2
View File
@@ -12,7 +12,7 @@ describe('extractEmbeddedImages', () => {
it('lifts a bare data:image URL out of prose', () => {
const result = extractEmbeddedImages(`describe this ${SAMPLE_PNG_DATA_URL}`)
expect(result.cleanedText).toBe('describe this')
expect(result.cleanedText).toBe('describe this ')
expect(result.images).toEqual([SAMPLE_PNG_DATA_URL])
})
@@ -79,7 +79,7 @@ describe('extractImageRefs', () => {
// lifted ref already renders that same attachment.
const result = extractImageRefs('@image:/tmp/cat.png\nwhat is in this photo?\n[screenshot]')
expect(result).toEqual({ cleanedText: 'what is in this photo?', refs: ['@image:/tmp/cat.png'] })
expect(result).toEqual({ cleanedText: 'what is in this photo?\n', refs: ['@image:/tmp/cat.png'] })
})
it('keeps [screenshot] when the message carries no image refs', () => {
+2 -2
View File
@@ -143,7 +143,7 @@ export function extractEmbeddedImages(text: string): EmbeddedImageExtraction {
pieces.push(text.slice(appendCursor))
return { cleanedText: pieces.join('').trim(), images }
return { cleanedText: pieces.join(''), images }
}
export function embeddedImageUrls(text: string): string[] {
@@ -193,5 +193,5 @@ export function extractImageRefs(text: string): { cleanedText: string; refs: str
cleanedText = cleanedText.replace(SCREENSHOT_PLACEHOLDER_LINE_RE, '')
}
return { cleanedText: cleanedText.trim(), refs }
return { cleanedText, refs }
}
@@ -30,12 +30,12 @@ describe('stripGeneratedImageEchoes', () => {
stripGeneratedImageEchoes('Here you go.\n\n![Generated image](https://cdn.example/cat.png)', [
'https://cdn.example/cat.png'
])
).toBe('Here you go.')
).toBe('Here you go.\n\n')
})
it('removes media links for generated local image paths', () => {
expect(stripGeneratedImageEchoes('Saved image: [Image: cat.png](#media:%2Ftmp%2Fcat.png)', ['/tmp/cat.png'])).toBe(
'Saved image:'
'Saved image: '
)
})
})
@@ -71,7 +71,7 @@ describe('dedupeGeneratedImageEchoesInParts', () => {
}
])
).toEqual([
{ text: 'Here is your peacock! Enjoy.', type: 'text' },
{ text: 'Here is your peacock! Enjoy.', type: 'text' },
{
result: { host_image: '/host/p.png', image: '/host/p.png', success: true },
toolName: 'image_generate',
+3 -3
View File
@@ -82,16 +82,16 @@ export function stripGeneratedImageEchoes(text: string, sources: readonly string
return text
}
// Join only the gap left by an image, never normalize unrelated Markdown or code.
// Remove only attachment spans; surrounding whitespace may be Markdown syntax.
let next = text
.replace(/([ \t]?)!\[[^\]\n]*\]\([^)\n]*\)([ \t]?)/g, (_match, before: string, after: string) => before || after)
.replace(/!\[[^\]\n]*\]\([^)\n]*\)/g, '')
.replace(/\[[^\]\n]*\]\(\s*#media:[^)\n]*\)/g, '')
for (const source of unique([...sources])) {
next = next.replace(new RegExp(String.raw`(^|[\s([{])<?${regexEscape(source)}>?(?=$|[\s)\]},.!?])`, 'g'), '$1')
}
return next.trim()
return next
}
/** Strip generated-image echoes from text parts, dropping any part left empty.
+2 -22
View File
@@ -12,7 +12,7 @@ const PREVIEW_MARKER_RE = /\[Preview:[^\]]+\]\(#preview[:/][^)]+\)/gi
const FENCE_LINE_RE = /^([ \t]*)(`{3,}|~{3,})([^\n]*)$/
const EMPTY_FENCE_BLOCK_RE = /(^|\n)[ \t]*(?:`{3,}|~{3,})[^\n]*\n[ \t]*(?:`{3,}|~{3,})[ \t]*(?=\n|$)/g
const CODE_FENCE_SPLIT_RE = /((?:```|~~~)[\s\S]*?(?:```|~~~))/g
const CODE_FENCE_SPLIT_RE = /((?:```|~~~)[\s\S]*?(?:```|~~~|$))/g
const INLINE_CODE_SPLIT_RE = /(`[^`\n]+`)/g
// Math spans as remark-math will see them: a `$$…$$` block, which may span
// lines, or a same-line `$…$`. A delimiter escaped as `\$` is prose — that is
@@ -637,31 +637,11 @@ export function preprocessMarkdown(text: string): string {
return part
}
// Whitespace-only segments (e.g. the `\n\n` between two adjacent
// fences) must NOT go through stripPreviewTargets — its internal
// .trim() would collapse them to '' and glue the surrounding
// fences together, producing things like ``````math which the
// markdown parser then reads as a single 6-backtick block.
if (!part.trim()) {
return part
}
// Preserve leading/trailing whitespace around the prose body so
// that fence-prose-fence sequences keep their blank-line gaps.
// stripPreviewTargets internally calls .trim() on its result for
// the benefit of its other (single-segment) callers; here we're
// operating on a SEGMENT of a larger document where outer
// whitespace is structural and must survive.
const leading = part.match(/^\s*/)?.[0] ?? ''
const trailing = part.match(/\s*$/)?.[0] ?? ''
// Run only on prose segments so `$5` literals and `\(` inside code
// blocks stay intact. The HTML-depth clamp belongs here for the same
// reason: a fenced block renders as code and never reaches rehype-raw,
// so escaping tags inside one would corrupt the listing for nothing.
const transformed = clampHtmlNestingDepth(normalizeVisibleProse(stripPreviewTargets(normalizeProseMath(part))))
return leading + transformed + trailing
return clampHtmlNestingDepth(normalizeVisibleProse(stripPreviewTargets(normalizeProseMath(part))))
})
.join('')
}
@@ -1,25 +1,39 @@
import { describe, expect, it } from 'vitest'
import { assistantTextPart, renderMediaTags } from './chat-messages/parts'
import { extractEmbeddedImages } from './embedded-images'
import { extractEmbeddedImages, extractImageRefs } from './embedded-images'
import { stripGeneratedImageEchoes } from './generated-images'
import { preprocessMarkdown } from './markdown-preprocess'
import { stripPreviewTargets } from './preview-targets'
const samples = [
'First line \nSecond line',
'Soft first\nSoft second',
' value = 1\n\n',
'```python\nvalue = 1 ',
'```python\n\nvalue = 1 \nvalue = 2 \n\n\n```'
]
describe('Markdown whitespace semantics', () => {
it('preserves hard and soft breaks and fenced whitespace through display preprocessing', () => {
for (const input of ['First line \nSecond line', 'Soft first\nSoft second', '```python\nvalue = 1 \nvalue = 2 \n```']) {
it('preserves significant whitespace through display preprocessing', () => {
for (const input of samples) {
expect(preprocessMarkdown(input)).toBe(input)
expect(preprocessMarkdown('\n\n' + input)).toBe('\n\n' + input)
}
})
it('preserves prose hard breaks when media and preview extraction runs', () => {
const prose = 'First line \nSecond line\n\n```python\nvalue = 1 \nvalue = 2 \n\n\nvalue = 3\n```'
it('preserves text outside removed media and preview spans', () => {
const image = 'data:image/png;base64,' + 'A'.repeat(64)
expect(stripPreviewTargets(prose + '\n\n[Preview: x](#preview:test)')).toContain(prose)
expect(renderMediaTags(prose + '\n\nMEDIA: /tmp/image.png')).toContain(prose)
expect(assistantTextPart(prose)).toMatchObject({ type: 'text', text: prose })
expect(extractEmbeddedImages(prose + '\n\n' + image).cleanedText).toContain(prose)
expect(stripGeneratedImageEchoes(prose + '\n\n![result](/tmp/image.png)', ['/tmp/image.png'])).toContain(prose)
for (const input of samples) {
expect(assistantTextPart(input)).toMatchObject({ type: 'text', text: input })
expect(renderMediaTags(input)).toBe(input)
// Prefix attachments so the unfinished fence remains the document end.
expect(stripPreviewTargets('[Preview: x](#preview:test)' + input)).toBe(input)
expect(extractEmbeddedImages(image + '\n' + input).cleanedText).toBe('\n' + input)
expect(stripGeneratedImageEchoes('![result](/tmp/image.png)\n' + input, ['/tmp/image.png'])).toBe('\n' + input)
expect(extractImageRefs('@image:/tmp/image.png\n' + input).cleanedText).toBe(input)
expect(extractEmbeddedImages(input + '\n\n' + image).cleanedText).toBe(input + '\n\n')
expect(stripGeneratedImageEchoes(input + '\n\n![result](/tmp/image.png)', ['/tmp/image.png'])).toBe(input + '\n\n')
}
})
})
+1 -1
View File
@@ -22,6 +22,6 @@ describe('preview target detection', () => {
expect(stripPreviewTargets('ready\n/tmp/mycelium-bunnies.html\nopen it')).toBe(
'ready\n/tmp/mycelium-bunnies.html\nopen it'
)
expect(stripPreviewTargets('[Preview: demo.html](#preview:%2Ftmp%2Fdemo.html)\nopen it')).toBe('open it')
expect(stripPreviewTargets('[Preview: demo.html](#preview:%2Ftmp%2Fdemo.html)\nopen it')).toBe('\nopen it')
})
})
+1 -1
View File
@@ -1,7 +1,7 @@
const PREVIEW_MARKDOWN_RE = /\[Preview:[^\]]+\]\((?<href>#preview[:/][^)]+)\)/gi
export function stripPreviewTargets(text: string): string {
return text.replace(PREVIEW_MARKDOWN_RE, '').trim()
return text.replace(PREVIEW_MARKDOWN_RE, '')
}
export function extractPreviewTargets(text: string): string[] {
+1 -1
View File
@@ -41,7 +41,7 @@ The desktop app is organized as a chat-first window with a left sidebar for navi
The center of the app. You get:
- **Streaming responses** with live tool activity and structured tool-call summaries as the agent works.
- **Markdown line breaks** follow Markdown semantics: two trailing spaces create a hard line break; an ordinary newline stays a soft break. Media and preview extraction preserve these breaks and whitespace inside fenced code blocks.
- **Markdown line breaks** follow Markdown semantics: two trailing spaces create a hard line break; an ordinary newline stays a soft break. Media and preview extraction preserve text outside removed attachment spans, including first-line code indentation and unfinished fenced-code spacing. Code display and Copy preserve leading blank lines, trailing spaces, and terminal blank lines from the Markdown parser.
- **The same conversation history** as every other Hermes surface — sessions started here resume in the CLI/TUI and vice versa.
- **Drag-and-drop files** anywhere in the chat area to attach them to your next message.
- **A right-hand preview rail** — render web pages, files, and tool outputs side by side while you keep chatting.