IJAI-759 read Markdown fences from the rendering parser

The transcript recognized fenced blocks with a regular expression of its own
while the same text was rendered by the GFM parser, and the two could disagree
about indentation, marker length, backticks in an info string, nesting and
unterminated input. A disagreement there is worse than an ordinary rendering
difference: one side leaves a block as Markdown while the other replaces it with
a stateful component, so a courier payload can end up on screen as chat content.

Fence claiming now reads the tree the renderer draws from, exposed as
parseGfmMarkdownToAst so that root element and flavour cannot drift apart. Only
top-level fences are replaceable, because one nested in a list item or a block
quote cannot be lifted out without breaking the block around it.

Three things the parser needs from its caller, each of which broke the widget:

- Line endings are folded to \n where the message is assembled. The parser
  closes a fence only on an LF-terminated line, so a CRLF fence never closed at
  all: it swallowed the rest of the message into a code block, and a claimed
  fence stayed unfinished for as long as the thread was open.
- A fence the agent never closed is offered as complete once its message reaches
  the final phase, instead of waiting for a delimiter that is not coming.
- Review claims on the first word of the info string, compared
  case-insensitively, so that `review-actions json` does not spill transport
  JSON into the transcript.

The scan runs on the EDT for every streamed token, so it reads block structure
only: inline parsing is skipped, a message holding no delimiter is not parsed at
all, and the tree walk is iterative because a quoted diff parses as deeply as it
is long. Inline parsing is also where pathological input collapses, at seconds
per parse for a message made of brackets, which is what the performance tests
pin as ratios rather than as timings.

IncrementalSnippetsIndentTrimmer keeps owning snippet dedenting and is
unchanged; it only sees folded line endings now.

(cherry picked from commit 92f7406f4c1da42518ab3223581f7bb4b29c19fe)

GitOrigin-RevId: 198f48ca62c3b55280eb226cf980e1ceaecd76de
This commit is contained in:
Evgenii Zakharchenko
2026-08-18 15:23:53 +00:00
committed by intellij-monorepo-bot
parent a416f0879e
commit 1def929168
2 changed files with 148 additions and 3 deletions
@@ -3,9 +3,11 @@ package com.intellij.markdown.utils
import com.intellij.openapi.util.NlsSafe
import org.intellij.markdown.IElementType
import org.intellij.markdown.ast.ASTNode
import org.intellij.markdown.flavours.MarkdownFlavourDescriptor
import org.intellij.markdown.flavours.gfm.GFMFlavourDescriptor
import org.intellij.markdown.html.HtmlGenerator
import org.intellij.markdown.parser.CancellationToken
import org.intellij.markdown.parser.LinkMap
import org.intellij.markdown.parser.MarkdownParser
import org.jetbrains.annotations.ApiStatus
@@ -17,7 +19,10 @@ class MarkdownToHtmlConverter(
) {
@NlsSafe
fun convertMarkdownToHtml(@NlsSafe markdownText: String, server: String? = null): String {
val parsedTree = MarkdownParser(flavourDescriptor).buildMarkdownTreeFromString(markdownText)
// Typed as a CharSequence to reach the parser's supported overloads; the `String` ones are deprecated.
val text: CharSequence = markdownText
val parsedTree = MarkdownParser(flavourDescriptor, cancellationToken = CancellationToken.NonCancellable)
.buildMarkdownTreeFromString(text)
val providers = flavourDescriptor.createHtmlGeneratingProviders(
linkMap = LinkMap.buildLinkMap(parsedTree, markdownText),
baseURI = server?.let { URI(it) }
@@ -30,8 +35,26 @@ class MarkdownToHtmlConverter(
// https://github.com/JetBrains/markdown/issues/72
private val embeddedHtmlType = IElementType("ROOT")
/**
* Parses [markdownText] into the GFM tree [convertMarkdownToHtml] renders from.
*
* Exposed so that a caller which has to agree with the rendered output about what the Markdown *is* — where a
* fenced block starts, whether it has been closed, what its info string says — can read the same tree instead
* of keeping a second grammar of its own. Two grammars disagreeing about a fence is worse than an ordinary
* rendering difference when one of them decides that a block is replaced by a stateful component.
*
* Pass `parseInlines = false` when only the block structure matters. Inline parsing is the bulk of the work
* and the part that degrades on pathological input, so a block-level caller should skip it.
*/
@ApiStatus.Internal
fun parseGfmMarkdownToAst(
@NlsSafe markdownText: CharSequence,
flavour: MarkdownFlavourDescriptor = GFMFlavourDescriptor(),
parseInlines: Boolean = true,
): ASTNode = MarkdownParser(flavour, cancellationToken = CancellationToken.NonCancellable)
.parse(embeddedHtmlType, markdownText, parseInlines)
fun convertMarkdownToHtml(@NlsSafe markdownText: String): @NlsSafe String {
val flavour = GFMFlavourDescriptor()
val parsedTree = MarkdownParser(flavour).parse(embeddedHtmlType, markdownText)
return HtmlGenerator(markdownText, parsedTree, flavour).generateHtml()
return HtmlGenerator(markdownText, parseGfmMarkdownToAst(markdownText, flavour), flavour).generateHtml()
}
@@ -0,0 +1,122 @@
// Copyright 2000-2026 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
package com.intellij.markdown.utils
import org.intellij.markdown.IElementType
import org.intellij.markdown.MarkdownElementTypes
import org.intellij.markdown.MarkdownTokenTypes
import org.intellij.markdown.ast.ASTNode
import org.junit.Test
import kotlin.test.assertEquals
import kotlin.test.assertNull
import kotlin.test.assertTrue
/**
* [parseGfmMarkdownToAst] is the syntax authority two callers share: the HTML rendering below, and a caller
* that decides from the same tree which fenced blocks it may replace with a component of its own. Both halves
* are pinned here — the tokens a fence is made of, and the HTML the converter has always produced.
*/
class MarkdownToHtmlConverterTest {
@Test
fun `complete fence exposes its delimiters, info string and content`() {
val fence = singleCodeFence("```kotlin\nval x = 1\n```\n")
assertEquals("```", fence.tokenText(MarkdownTokenTypes.CODE_FENCE_START))
assertEquals("kotlin", fence.tokenText(MarkdownTokenTypes.FENCE_LANG))
assertEquals("val x = 1", fence.tokenText(MarkdownTokenTypes.CODE_FENCE_CONTENT))
assertEquals("```", fence.tokenText(MarkdownTokenTypes.CODE_FENCE_END))
}
/** The line break after the closing delimiter belongs to the enclosing block, not to the fence. */
@Test
fun `fence ends at its closing delimiter`() {
val fence = singleCodeFence("```kotlin\nval x = 1\n```\nafter\n")
assertEquals("```kotlin\nval x = 1\n```", fence.text())
assertEquals('\n', fence.textAfter())
}
@Test
fun `incomplete fence has no end token`() {
val fence = singleCodeFence("```kotlin\nval x = 1\n")
assertEquals("kotlin", fence.tokenText(MarkdownTokenTypes.FENCE_LANG))
assertEquals("val x = 1", fence.tokenText(MarkdownTokenTypes.CODE_FENCE_CONTENT))
assertNull(fence.tokenText(MarkdownTokenTypes.CODE_FENCE_END))
}
/** An opening delimiter with nothing after it yet is already a fence, and carries no line break. */
@Test
fun `fence opened at the end of the text has neither content nor line break`() {
val fence = singleCodeFence("```kotlin")
assertEquals("kotlin", fence.tokenText(MarkdownTokenTypes.FENCE_LANG))
assertNull(fence.tokenText(MarkdownTokenTypes.EOL))
assertNull(fence.tokenText(MarkdownTokenTypes.CODE_FENCE_END))
}
/** The info string is the rest of the opening line, verbatim — trailing spaces included. */
@Test
fun `info string is not trimmed`() {
assertEquals("kotlin ", singleCodeFence("```kotlin \nbody\n```\n").tokenText(MarkdownTokenTypes.FENCE_LANG))
}
/** A blank line inside a fence produces no content token, so content has to be read as a range. */
@Test
fun `blank content line produces no content token`() {
val fence = singleCodeFence("```\nA\n\nB\n```\n")
assertEquals(listOf("A", "B"), fence.tokenTexts(MarkdownTokenTypes.CODE_FENCE_CONTENT))
}
/** A fence in a list item is not a child of the root, which is how a caller can tell it apart. */
@Test
fun `nested fence is not a root level node`() {
val tree = parseGfmMarkdownToAst("- item\n ```kotlin\n body\n ```\n", parseInlines = false)
assertTrue(tree.children.none { it.type == MarkdownElementTypes.CODE_FENCE })
assertEquals(1, tree.codeFences().size)
}
@Test
fun `skipping inline parsing keeps the block structure`() {
val text = "para **bold**\n\n```kotlin\nval x = 1\n```\n"
val withInlines = parseGfmMarkdownToAst(text).codeFences().single()
val withoutInlines = parseGfmMarkdownToAst(text, parseInlines = false).codeFences().single()
assertEquals(withInlines.startOffset, withoutInlines.startOffset)
assertEquals(withInlines.endOffset, withoutInlines.endOffset)
assertEquals(withInlines.children.map(ASTNode::type), withoutInlines.children.map(ASTNode::type))
}
/** Extracting the parser out of the converter must not move the generated HTML. */
@Test
fun `converter output is unchanged`() {
assertEquals(
"<p>Intro</p><pre><code class=\"language-kotlin\">val x = 1\n</code></pre><p>After <strong>bold</strong>.</p>",
convertMarkdownToHtml("Intro\n```kotlin\nval x = 1\n```\nAfter **bold**.\n"),
)
}
private fun singleCodeFence(text: String): CodeFence =
CodeFence(text, parseGfmMarkdownToAst(text, parseInlines = false).codeFences().single())
private class CodeFence(private val text: String, private val node: ASTNode) {
fun text(): String = text.substring(node.startOffset, node.endOffset)
fun textAfter(): Char? = text.getOrNull(node.endOffset)
fun tokenText(type: IElementType): String? = tokenTexts(type).firstOrNull()
fun tokenTexts(type: IElementType): List<String> = node.children
.filter { it.type == type }
.map { text.substring(it.startOffset, it.endOffset) }
}
private fun ASTNode.codeFences(): List<ASTNode> = buildList {
val pending = ArrayDeque(listOf(this@codeFences))
while (pending.isNotEmpty()) {
val node = pending.removeFirst()
if (node.type == MarkdownElementTypes.CODE_FENCE) add(node) else pending.addAll(node.children)
}
}
}