diff --git a/platform/markdown-utils/src/com/intellij/markdown/utils/MarkdownToHtmlConverter.kt b/platform/markdown-utils/src/com/intellij/markdown/utils/MarkdownToHtmlConverter.kt index 1617048eb170..a4f608596c2d 100644 --- a/platform/markdown-utils/src/com/intellij/markdown/utils/MarkdownToHtmlConverter.kt +++ b/platform/markdown-utils/src/com/intellij/markdown/utils/MarkdownToHtmlConverter.kt @@ -3,9 +3,11 @@ package com.intellij.markdown.utils import com.intellij.openapi.util.NlsSafe import org.intellij.markdown.IElementType +import org.intellij.markdown.ast.ASTNode import org.intellij.markdown.flavours.MarkdownFlavourDescriptor import org.intellij.markdown.flavours.gfm.GFMFlavourDescriptor import org.intellij.markdown.html.HtmlGenerator +import org.intellij.markdown.parser.CancellationToken import org.intellij.markdown.parser.LinkMap import org.intellij.markdown.parser.MarkdownParser import org.jetbrains.annotations.ApiStatus @@ -17,7 +19,10 @@ class MarkdownToHtmlConverter( ) { @NlsSafe fun convertMarkdownToHtml(@NlsSafe markdownText: String, server: String? = null): String { - val parsedTree = MarkdownParser(flavourDescriptor).buildMarkdownTreeFromString(markdownText) + // Typed as a CharSequence to reach the parser's supported overloads; the `String` ones are deprecated. + val text: CharSequence = markdownText + val parsedTree = MarkdownParser(flavourDescriptor, cancellationToken = CancellationToken.NonCancellable) + .buildMarkdownTreeFromString(text) val providers = flavourDescriptor.createHtmlGeneratingProviders( linkMap = LinkMap.buildLinkMap(parsedTree, markdownText), baseURI = server?.let { URI(it) } @@ -30,8 +35,26 @@ class MarkdownToHtmlConverter( // https://github.com/JetBrains/markdown/issues/72 private val embeddedHtmlType = IElementType("ROOT") +/** + * Parses [markdownText] into the GFM tree [convertMarkdownToHtml] renders from. + * + * Exposed so that a caller which has to agree with the rendered output about what the Markdown *is* — where a + * fenced block starts, whether it has been closed, what its info string says — can read the same tree instead + * of keeping a second grammar of its own. Two grammars disagreeing about a fence is worse than an ordinary + * rendering difference when one of them decides that a block is replaced by a stateful component. + * + * Pass `parseInlines = false` when only the block structure matters. Inline parsing is the bulk of the work + * and the part that degrades on pathological input, so a block-level caller should skip it. + */ +@ApiStatus.Internal +fun parseGfmMarkdownToAst( + @NlsSafe markdownText: CharSequence, + flavour: MarkdownFlavourDescriptor = GFMFlavourDescriptor(), + parseInlines: Boolean = true, +): ASTNode = MarkdownParser(flavour, cancellationToken = CancellationToken.NonCancellable) + .parse(embeddedHtmlType, markdownText, parseInlines) + fun convertMarkdownToHtml(@NlsSafe markdownText: String): @NlsSafe String { val flavour = GFMFlavourDescriptor() - val parsedTree = MarkdownParser(flavour).parse(embeddedHtmlType, markdownText) - return HtmlGenerator(markdownText, parsedTree, flavour).generateHtml() + return HtmlGenerator(markdownText, parseGfmMarkdownToAst(markdownText, flavour), flavour).generateHtml() } \ No newline at end of file diff --git a/platform/markdown-utils/test/com/intellij/markdown/utils/MarkdownToHtmlConverterTest.kt b/platform/markdown-utils/test/com/intellij/markdown/utils/MarkdownToHtmlConverterTest.kt new file mode 100644 index 000000000000..0f80f7965c59 --- /dev/null +++ b/platform/markdown-utils/test/com/intellij/markdown/utils/MarkdownToHtmlConverterTest.kt @@ -0,0 +1,122 @@ +// Copyright 2000-2026 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license. +package com.intellij.markdown.utils + +import org.intellij.markdown.IElementType +import org.intellij.markdown.MarkdownElementTypes +import org.intellij.markdown.MarkdownTokenTypes +import org.intellij.markdown.ast.ASTNode +import org.junit.Test +import kotlin.test.assertEquals +import kotlin.test.assertNull +import kotlin.test.assertTrue + +/** + * [parseGfmMarkdownToAst] is the syntax authority two callers share: the HTML rendering below, and a caller + * that decides from the same tree which fenced blocks it may replace with a component of its own. Both halves + * are pinned here — the tokens a fence is made of, and the HTML the converter has always produced. + */ +class MarkdownToHtmlConverterTest { + @Test + fun `complete fence exposes its delimiters, info string and content`() { + val fence = singleCodeFence("```kotlin\nval x = 1\n```\n") + + assertEquals("```", fence.tokenText(MarkdownTokenTypes.CODE_FENCE_START)) + assertEquals("kotlin", fence.tokenText(MarkdownTokenTypes.FENCE_LANG)) + assertEquals("val x = 1", fence.tokenText(MarkdownTokenTypes.CODE_FENCE_CONTENT)) + assertEquals("```", fence.tokenText(MarkdownTokenTypes.CODE_FENCE_END)) + } + + /** The line break after the closing delimiter belongs to the enclosing block, not to the fence. */ + @Test + fun `fence ends at its closing delimiter`() { + val fence = singleCodeFence("```kotlin\nval x = 1\n```\nafter\n") + + assertEquals("```kotlin\nval x = 1\n```", fence.text()) + assertEquals('\n', fence.textAfter()) + } + + @Test + fun `incomplete fence has no end token`() { + val fence = singleCodeFence("```kotlin\nval x = 1\n") + + assertEquals("kotlin", fence.tokenText(MarkdownTokenTypes.FENCE_LANG)) + assertEquals("val x = 1", fence.tokenText(MarkdownTokenTypes.CODE_FENCE_CONTENT)) + assertNull(fence.tokenText(MarkdownTokenTypes.CODE_FENCE_END)) + } + + /** An opening delimiter with nothing after it yet is already a fence, and carries no line break. */ + @Test + fun `fence opened at the end of the text has neither content nor line break`() { + val fence = singleCodeFence("```kotlin") + + assertEquals("kotlin", fence.tokenText(MarkdownTokenTypes.FENCE_LANG)) + assertNull(fence.tokenText(MarkdownTokenTypes.EOL)) + assertNull(fence.tokenText(MarkdownTokenTypes.CODE_FENCE_END)) + } + + /** The info string is the rest of the opening line, verbatim — trailing spaces included. */ + @Test + fun `info string is not trimmed`() { + assertEquals("kotlin ", singleCodeFence("```kotlin \nbody\n```\n").tokenText(MarkdownTokenTypes.FENCE_LANG)) + } + + /** A blank line inside a fence produces no content token, so content has to be read as a range. */ + @Test + fun `blank content line produces no content token`() { + val fence = singleCodeFence("```\nA\n\nB\n```\n") + + assertEquals(listOf("A", "B"), fence.tokenTexts(MarkdownTokenTypes.CODE_FENCE_CONTENT)) + } + + /** A fence in a list item is not a child of the root, which is how a caller can tell it apart. */ + @Test + fun `nested fence is not a root level node`() { + val tree = parseGfmMarkdownToAst("- item\n ```kotlin\n body\n ```\n", parseInlines = false) + + assertTrue(tree.children.none { it.type == MarkdownElementTypes.CODE_FENCE }) + assertEquals(1, tree.codeFences().size) + } + + @Test + fun `skipping inline parsing keeps the block structure`() { + val text = "para **bold**\n\n```kotlin\nval x = 1\n```\n" + val withInlines = parseGfmMarkdownToAst(text).codeFences().single() + val withoutInlines = parseGfmMarkdownToAst(text, parseInlines = false).codeFences().single() + + assertEquals(withInlines.startOffset, withoutInlines.startOffset) + assertEquals(withInlines.endOffset, withoutInlines.endOffset) + assertEquals(withInlines.children.map(ASTNode::type), withoutInlines.children.map(ASTNode::type)) + } + + /** Extracting the parser out of the converter must not move the generated HTML. */ + @Test + fun `converter output is unchanged`() { + assertEquals( + "

Intro

val x = 1\n

After bold.

", + convertMarkdownToHtml("Intro\n```kotlin\nval x = 1\n```\nAfter **bold**.\n"), + ) + } + + private fun singleCodeFence(text: String): CodeFence = + CodeFence(text, parseGfmMarkdownToAst(text, parseInlines = false).codeFences().single()) + + private class CodeFence(private val text: String, private val node: ASTNode) { + fun text(): String = text.substring(node.startOffset, node.endOffset) + + fun textAfter(): Char? = text.getOrNull(node.endOffset) + + fun tokenText(type: IElementType): String? = tokenTexts(type).firstOrNull() + + fun tokenTexts(type: IElementType): List = node.children + .filter { it.type == type } + .map { text.substring(it.startOffset, it.endOffset) } + } + + private fun ASTNode.codeFences(): List = buildList { + val pending = ArrayDeque(listOf(this@codeFences)) + while (pending.isNotEmpty()) { + val node = pending.removeFirst() + if (node.type == MarkdownElementTypes.CODE_FENCE) add(node) else pending.addAll(node.children) + } + } +}