mirror of
https://gitflic.ru/project/openide/openide.git
synced 2026-09-27 10:03:11 +07:00
IJAI-759 read Markdown fences from the rendering parser
The transcript recognized fenced blocks with a regular expression of its own while the same text was rendered by the GFM parser, and the two could disagree about indentation, marker length, backticks in an info string, nesting and unterminated input. A disagreement there is worse than an ordinary rendering difference: one side leaves a block as Markdown while the other replaces it with a stateful component, so a courier payload can end up on screen as chat content. Fence claiming now reads the tree the renderer draws from, exposed as parseGfmMarkdownToAst so that root element and flavour cannot drift apart. Only top-level fences are replaceable, because one nested in a list item or a block quote cannot be lifted out without breaking the block around it. Three things the parser needs from its caller, each of which broke the widget: - Line endings are folded to \n where the message is assembled. The parser closes a fence only on an LF-terminated line, so a CRLF fence never closed at all: it swallowed the rest of the message into a code block, and a claimed fence stayed unfinished for as long as the thread was open. - A fence the agent never closed is offered as complete once its message reaches the final phase, instead of waiting for a delimiter that is not coming. - Review claims on the first word of the info string, compared case-insensitively, so that `review-actions json` does not spill transport JSON into the transcript. The scan runs on the EDT for every streamed token, so it reads block structure only: inline parsing is skipped, a message holding no delimiter is not parsed at all, and the tree walk is iterative because a quoted diff parses as deeply as it is long. Inline parsing is also where pathological input collapses, at seconds per parse for a message made of brackets, which is what the performance tests pin as ratios rather than as timings. IncrementalSnippetsIndentTrimmer keeps owning snippet dedenting and is unchanged; it only sees folded line endings now. (cherry picked from commit 92f7406f4c1da42518ab3223581f7bb4b29c19fe) GitOrigin-RevId: 198f48ca62c3b55280eb226cf980e1ceaecd76de
This commit is contained in:
committed by
intellij-monorepo-bot
parent
a416f0879e
commit
1def929168
@@ -3,9 +3,11 @@ package com.intellij.markdown.utils
|
||||
|
||||
import com.intellij.openapi.util.NlsSafe
|
||||
import org.intellij.markdown.IElementType
|
||||
import org.intellij.markdown.ast.ASTNode
|
||||
import org.intellij.markdown.flavours.MarkdownFlavourDescriptor
|
||||
import org.intellij.markdown.flavours.gfm.GFMFlavourDescriptor
|
||||
import org.intellij.markdown.html.HtmlGenerator
|
||||
import org.intellij.markdown.parser.CancellationToken
|
||||
import org.intellij.markdown.parser.LinkMap
|
||||
import org.intellij.markdown.parser.MarkdownParser
|
||||
import org.jetbrains.annotations.ApiStatus
|
||||
@@ -17,7 +19,10 @@ class MarkdownToHtmlConverter(
|
||||
) {
|
||||
@NlsSafe
|
||||
fun convertMarkdownToHtml(@NlsSafe markdownText: String, server: String? = null): String {
|
||||
val parsedTree = MarkdownParser(flavourDescriptor).buildMarkdownTreeFromString(markdownText)
|
||||
// Typed as a CharSequence to reach the parser's supported overloads; the `String` ones are deprecated.
|
||||
val text: CharSequence = markdownText
|
||||
val parsedTree = MarkdownParser(flavourDescriptor, cancellationToken = CancellationToken.NonCancellable)
|
||||
.buildMarkdownTreeFromString(text)
|
||||
val providers = flavourDescriptor.createHtmlGeneratingProviders(
|
||||
linkMap = LinkMap.buildLinkMap(parsedTree, markdownText),
|
||||
baseURI = server?.let { URI(it) }
|
||||
@@ -30,8 +35,26 @@ class MarkdownToHtmlConverter(
|
||||
// https://github.com/JetBrains/markdown/issues/72
|
||||
private val embeddedHtmlType = IElementType("ROOT")
|
||||
|
||||
/**
|
||||
* Parses [markdownText] into the GFM tree [convertMarkdownToHtml] renders from.
|
||||
*
|
||||
* Exposed so that a caller which has to agree with the rendered output about what the Markdown *is* — where a
|
||||
* fenced block starts, whether it has been closed, what its info string says — can read the same tree instead
|
||||
* of keeping a second grammar of its own. Two grammars disagreeing about a fence is worse than an ordinary
|
||||
* rendering difference when one of them decides that a block is replaced by a stateful component.
|
||||
*
|
||||
* Pass `parseInlines = false` when only the block structure matters. Inline parsing is the bulk of the work
|
||||
* and the part that degrades on pathological input, so a block-level caller should skip it.
|
||||
*/
|
||||
@ApiStatus.Internal
|
||||
fun parseGfmMarkdownToAst(
|
||||
@NlsSafe markdownText: CharSequence,
|
||||
flavour: MarkdownFlavourDescriptor = GFMFlavourDescriptor(),
|
||||
parseInlines: Boolean = true,
|
||||
): ASTNode = MarkdownParser(flavour, cancellationToken = CancellationToken.NonCancellable)
|
||||
.parse(embeddedHtmlType, markdownText, parseInlines)
|
||||
|
||||
fun convertMarkdownToHtml(@NlsSafe markdownText: String): @NlsSafe String {
|
||||
val flavour = GFMFlavourDescriptor()
|
||||
val parsedTree = MarkdownParser(flavour).parse(embeddedHtmlType, markdownText)
|
||||
return HtmlGenerator(markdownText, parsedTree, flavour).generateHtml()
|
||||
return HtmlGenerator(markdownText, parseGfmMarkdownToAst(markdownText, flavour), flavour).generateHtml()
|
||||
}
|
||||
+122
@@ -0,0 +1,122 @@
|
||||
// Copyright 2000-2026 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
package com.intellij.markdown.utils
|
||||
|
||||
import org.intellij.markdown.IElementType
|
||||
import org.intellij.markdown.MarkdownElementTypes
|
||||
import org.intellij.markdown.MarkdownTokenTypes
|
||||
import org.intellij.markdown.ast.ASTNode
|
||||
import org.junit.Test
|
||||
import kotlin.test.assertEquals
|
||||
import kotlin.test.assertNull
|
||||
import kotlin.test.assertTrue
|
||||
|
||||
/**
|
||||
* [parseGfmMarkdownToAst] is the syntax authority two callers share: the HTML rendering below, and a caller
|
||||
* that decides from the same tree which fenced blocks it may replace with a component of its own. Both halves
|
||||
* are pinned here — the tokens a fence is made of, and the HTML the converter has always produced.
|
||||
*/
|
||||
class MarkdownToHtmlConverterTest {
|
||||
@Test
|
||||
fun `complete fence exposes its delimiters, info string and content`() {
|
||||
val fence = singleCodeFence("```kotlin\nval x = 1\n```\n")
|
||||
|
||||
assertEquals("```", fence.tokenText(MarkdownTokenTypes.CODE_FENCE_START))
|
||||
assertEquals("kotlin", fence.tokenText(MarkdownTokenTypes.FENCE_LANG))
|
||||
assertEquals("val x = 1", fence.tokenText(MarkdownTokenTypes.CODE_FENCE_CONTENT))
|
||||
assertEquals("```", fence.tokenText(MarkdownTokenTypes.CODE_FENCE_END))
|
||||
}
|
||||
|
||||
/** The line break after the closing delimiter belongs to the enclosing block, not to the fence. */
|
||||
@Test
|
||||
fun `fence ends at its closing delimiter`() {
|
||||
val fence = singleCodeFence("```kotlin\nval x = 1\n```\nafter\n")
|
||||
|
||||
assertEquals("```kotlin\nval x = 1\n```", fence.text())
|
||||
assertEquals('\n', fence.textAfter())
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `incomplete fence has no end token`() {
|
||||
val fence = singleCodeFence("```kotlin\nval x = 1\n")
|
||||
|
||||
assertEquals("kotlin", fence.tokenText(MarkdownTokenTypes.FENCE_LANG))
|
||||
assertEquals("val x = 1", fence.tokenText(MarkdownTokenTypes.CODE_FENCE_CONTENT))
|
||||
assertNull(fence.tokenText(MarkdownTokenTypes.CODE_FENCE_END))
|
||||
}
|
||||
|
||||
/** An opening delimiter with nothing after it yet is already a fence, and carries no line break. */
|
||||
@Test
|
||||
fun `fence opened at the end of the text has neither content nor line break`() {
|
||||
val fence = singleCodeFence("```kotlin")
|
||||
|
||||
assertEquals("kotlin", fence.tokenText(MarkdownTokenTypes.FENCE_LANG))
|
||||
assertNull(fence.tokenText(MarkdownTokenTypes.EOL))
|
||||
assertNull(fence.tokenText(MarkdownTokenTypes.CODE_FENCE_END))
|
||||
}
|
||||
|
||||
/** The info string is the rest of the opening line, verbatim — trailing spaces included. */
|
||||
@Test
|
||||
fun `info string is not trimmed`() {
|
||||
assertEquals("kotlin ", singleCodeFence("```kotlin \nbody\n```\n").tokenText(MarkdownTokenTypes.FENCE_LANG))
|
||||
}
|
||||
|
||||
/** A blank line inside a fence produces no content token, so content has to be read as a range. */
|
||||
@Test
|
||||
fun `blank content line produces no content token`() {
|
||||
val fence = singleCodeFence("```\nA\n\nB\n```\n")
|
||||
|
||||
assertEquals(listOf("A", "B"), fence.tokenTexts(MarkdownTokenTypes.CODE_FENCE_CONTENT))
|
||||
}
|
||||
|
||||
/** A fence in a list item is not a child of the root, which is how a caller can tell it apart. */
|
||||
@Test
|
||||
fun `nested fence is not a root level node`() {
|
||||
val tree = parseGfmMarkdownToAst("- item\n ```kotlin\n body\n ```\n", parseInlines = false)
|
||||
|
||||
assertTrue(tree.children.none { it.type == MarkdownElementTypes.CODE_FENCE })
|
||||
assertEquals(1, tree.codeFences().size)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `skipping inline parsing keeps the block structure`() {
|
||||
val text = "para **bold**\n\n```kotlin\nval x = 1\n```\n"
|
||||
val withInlines = parseGfmMarkdownToAst(text).codeFences().single()
|
||||
val withoutInlines = parseGfmMarkdownToAst(text, parseInlines = false).codeFences().single()
|
||||
|
||||
assertEquals(withInlines.startOffset, withoutInlines.startOffset)
|
||||
assertEquals(withInlines.endOffset, withoutInlines.endOffset)
|
||||
assertEquals(withInlines.children.map(ASTNode::type), withoutInlines.children.map(ASTNode::type))
|
||||
}
|
||||
|
||||
/** Extracting the parser out of the converter must not move the generated HTML. */
|
||||
@Test
|
||||
fun `converter output is unchanged`() {
|
||||
assertEquals(
|
||||
"<p>Intro</p><pre><code class=\"language-kotlin\">val x = 1\n</code></pre><p>After <strong>bold</strong>.</p>",
|
||||
convertMarkdownToHtml("Intro\n```kotlin\nval x = 1\n```\nAfter **bold**.\n"),
|
||||
)
|
||||
}
|
||||
|
||||
private fun singleCodeFence(text: String): CodeFence =
|
||||
CodeFence(text, parseGfmMarkdownToAst(text, parseInlines = false).codeFences().single())
|
||||
|
||||
private class CodeFence(private val text: String, private val node: ASTNode) {
|
||||
fun text(): String = text.substring(node.startOffset, node.endOffset)
|
||||
|
||||
fun textAfter(): Char? = text.getOrNull(node.endOffset)
|
||||
|
||||
fun tokenText(type: IElementType): String? = tokenTexts(type).firstOrNull()
|
||||
|
||||
fun tokenTexts(type: IElementType): List<String> = node.children
|
||||
.filter { it.type == type }
|
||||
.map { text.substring(it.startOffset, it.endOffset) }
|
||||
}
|
||||
|
||||
private fun ASTNode.codeFences(): List<ASTNode> = buildList {
|
||||
val pending = ArrayDeque(listOf(this@codeFences))
|
||||
while (pending.isNotEmpty()) {
|
||||
val node = pending.removeFirst()
|
||||
if (node.type == MarkdownElementTypes.CODE_FENCE) add(node) else pending.addAll(node.children)
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user