[grazie] Update TextExtractor's Javadoc and add guard checks

Merge-request: IJ-MR-174142
Merged-by: Ilia Permiashkin <ilia.permiashkin@jetbrains.com>

GitOrigin-RevId: db257a78238b2e315b11f708a02b1f08de024fef
This commit is contained in:
Ilia Permiashkin
2025-09-02 13:10:37 +00:00
committed by intellij-monorepo-bot
parent b286322ddd
commit 7aed5175ec
6 changed files with 47 additions and 22 deletions
@@ -2,6 +2,7 @@
package com.intellij.grazie.ide.language.json
import com.intellij.grazie.text.*
import com.intellij.grazie.text.TextContent.TextDomain
import com.intellij.grazie.utils.replaceBackslashEscapes
import com.intellij.json.JsonSpellcheckerStrategy.JsonSchemaSpellcheckerClientForJson
import com.intellij.json.psi.JsonStringLiteral
@@ -9,10 +10,11 @@ import com.intellij.psi.PsiComment
import com.intellij.psi.PsiElement
class JsonTextExtractor : TextExtractor() {
override fun buildTextContent(element: PsiElement, allowedDomains: MutableSet<TextContent.TextDomain>): TextContent? {
if (element is JsonStringLiteral && JsonSchemaSpellcheckerClientForJson(element).matchesNameFromSchema()) return null
override fun buildTextContent(element: PsiElement, allowedDomains: MutableSet<TextDomain>): TextContent? {
if (element is PsiComment || element is JsonStringLiteral) {
val domain = if (element is PsiComment) TextContent.TextDomain.COMMENTS else TextContent.TextDomain.LITERALS
val domain = if (element is PsiComment) TextDomain.COMMENTS else TextDomain.LITERALS
if (domain !in allowedDomains) return null
if (element is JsonStringLiteral && JsonSchemaSpellcheckerClientForJson(element).matchesNameFromSchema()) return null
val content = TextContentBuilder.FromPsi.build(element, domain) ?: return null
return content.replaceBackslashEscapes()
}
@@ -2,6 +2,7 @@ package com.intellij.grazie.ide.language.properties;
import com.intellij.grazie.text.TextContent;
import com.intellij.grazie.text.TextContent.Exclusion;
import com.intellij.grazie.text.TextContent.TextDomain;
import com.intellij.grazie.text.TextContentBuilder;
import com.intellij.grazie.text.TextExtractor;
import com.intellij.grazie.utils.HtmlUtilsKt;
@@ -34,16 +35,16 @@ final class PropertyTextExtractor extends TextExtractor {
private static final Pattern trailingSlash = Pattern.compile("\\\\\n");
@Override
protected @NotNull List<TextContent> buildTextContents(@NotNull PsiElement root, @NotNull Set<TextContent.TextDomain> allowedDomains) {
if (root instanceof PsiComment) {
protected @NotNull List<TextContent> buildTextContents(@NotNull PsiElement root, @NotNull Set<TextDomain> allowedDomains) {
if (root instanceof PsiComment && allowedDomains.contains(COMMENTS)) {
List<PsiElement> roots = PsiUtilsKt.getNotSoDistantSimilarSiblings(root, e ->
PropertiesTokenTypes.COMMENTS.contains(PsiUtilCore.getElementType(e)));
return ContainerUtil.createMaybeSingletonList(
TextContent.joinWithWhitespace('\n', ContainerUtil.mapNotNull(roots, c ->
TextContentBuilder.FromPsi.removingIndents(" \t#!").build(c, COMMENTS))));
}
if (PsiUtilCore.getElementType(root) == PropertiesTokenTypes.VALUE_CHARACTERS) {
TextContent content = TextContent.builder().build(root, TextContent.TextDomain.PLAIN_TEXT);
if (PsiUtilCore.getElementType(root) == PropertiesTokenTypes.VALUE_CHARACTERS && allowedDomains.contains(TextDomain.PLAIN_TEXT)) {
TextContent content = TextContent.builder().build(root, TextDomain.PLAIN_TEXT);
if (content != null) {
content = content.excludeRanges(ContainerUtil.map(Text.allOccurrences(apostrophes, content), Exclusion::exclude));
content = content.excludeRanges(ContainerUtil.map(Text.allOccurrences(continuationIndent, content), Exclusion::exclude));
@@ -1,5 +1,6 @@
package com.intellij.grazie.text;
import com.intellij.grazie.text.TextContent.TextDomain;
import com.intellij.grazie.utils.Text;
import com.intellij.openapi.util.TextRange;
import com.intellij.psi.PsiElement;
@@ -14,19 +15,21 @@ import java.util.List;
import java.util.Set;
import java.util.regex.Pattern;
import static com.intellij.grazie.text.TextContent.TextDomain.PLAIN_TEXT;
public class PlainTextExtractor extends TextExtractor {
private static final Pattern paragraphEnd = Pattern.compile("\\n\\s*?\\n\\s*");
@Override
protected @NotNull List<TextContent> buildTextContents(@NotNull PsiElement root, @NotNull Set<TextContent.TextDomain> allowedDomains) {
if (root instanceof PsiPlainText && root.getContainingFile().getName().endsWith(".txt")) {
protected @NotNull List<TextContent> buildTextContents(@NotNull PsiElement root, @NotNull Set<TextDomain> allowedDomains) {
if (root instanceof PsiPlainText && root.getContainingFile().getName().endsWith(".txt") && allowedDomains.contains(PLAIN_TEXT)) {
String text = root.getText();
List<TextContent> result = new ArrayList<>();
int[] ends = StreamEx.of(Text.allOccurrences(paragraphEnd, text)).mapToInt(TextRange::getStartOffset).append(text.length()).toArray();
for (int i = 0; i < ends.length; i++) {
int start = i == 0 ? 0 : ends[i - 1];
int end = ends[i];
ContainerUtil.addIfNotNull(result, TextContent.builder().build(root, TextContent.TextDomain.PLAIN_TEXT, new TextRange(start, end)));
ContainerUtil.addIfNotNull(result, TextContent.builder().build(root, PLAIN_TEXT, new TextRange(start, end)));
}
return result;
}
@@ -49,12 +49,29 @@ public abstract class TextExtractor {
/**
* Extract text from the given PSI element, if possible.
* The returned text is most often fully embedded in {@code element},
* but it may also include other PSI elements (e.g. adjacent comments).
* but it may also include other PSI elements (e.g., adjacent comments).
* In the latter case, this extension should return an equal {@link TextContent} for every one of those adjacent elements.
* <p>
* Typical usage:
*
* <pre><code class="java">TextContentBuilder.FromPsi.build(element, textDomain)</code></pre>
*
* Implementation guidance:
* <p>
* To maximize performance, guard against unnecessary (and sometimes quite expensive) operations by checking that
* the requested textDomain is contained in allowedDomains before extracting.
*
* <pre><code class="java">
* if (shouldExtractTextContent(root) && allowedDomains.contains(textDomain)) {
* // some other potentially performance-intensive operations
* return TextContentBuilder.FromPsi.build(root, textDomain)
* }
* </code></pre>
*
* See concrete implementations (e.g., in ChatInputTextExtractor, JsonTextExtractor, GoTextExtractor, etc.) for
* examples.
* @param allowedDomains the set of the text domains that are expected by the caller.
* The extension may check this set before doing unnecessary expensive PSI traversal
* to improve the performance,
* but it's not necessary.
* @see TextContentBuilder
* @see #buildTextContents
*/
@@ -2,6 +2,7 @@
package com.intellij.grazie.ide.language.yaml
import com.intellij.grazie.text.TextContent
import com.intellij.grazie.text.TextContent.TextDomain
import com.intellij.grazie.text.TextContentBuilder
import com.intellij.grazie.text.TextExtractor
import com.intellij.grazie.utils.getNotSoDistantSimilarSiblings
@@ -17,19 +18,19 @@ import org.jetbrains.yaml.psi.impl.YAMLAnchorImpl
private class YamlTextExtractor : TextExtractor() {
private val commentBuilder = TextContentBuilder.FromPsi.removingIndents(" \t#")
override fun buildTextContent(root: PsiElement, allowedDomains: MutableSet<TextContent.TextDomain>): TextContent? {
if (root is PsiComment) {
override fun buildTextContent(root: PsiElement, allowedDomains: MutableSet<TextDomain>): TextContent? {
if (TextDomain.COMMENTS in allowedDomains && root is PsiComment) {
val siblings = getNotSoDistantSimilarSiblings(root, TokenSet.create(WHITESPACE, INDENT, EOL)) { it.elementType == COMMENT }
return TextContent.joinWithWhitespace('\n', siblings.mapNotNull { commentBuilder.build(it, TextContent.TextDomain.COMMENTS) })
return TextContent.joinWithWhitespace('\n', siblings.mapNotNull { commentBuilder.build(it, TextDomain.COMMENTS) })
}
if (root is YAMLScalar || (root.node != null && root.node.elementType == SCALAR_KEY)) {
if (TextDomain.LITERALS in allowedDomains && (root is YAMLScalar || (root.node != null && root.node.elementType == SCALAR_KEY))) {
if (JsonSchemaSpellcheckerClientForYaml(root).matchesNameFromSchema()) {
return null
}
return TextContentBuilder.FromPsi.excluding { isStealth(it) }.build(root, TextContent.TextDomain.LITERALS)
return TextContentBuilder.FromPsi.excluding { isStealth(it) }.build(root, TextDomain.LITERALS)
}
if (root is YAMLAnchorImpl && root.parent !is YAMLScalar) {
return TextContentBuilder.FromPsi.excluding { isStealth(it) }.build(root, TextContent.TextDomain.LITERALS)
if (TextDomain.LITERALS in allowedDomains && (root is YAMLAnchorImpl && root.parent !is YAMLScalar)) {
return TextContentBuilder.FromPsi.excluding { isStealth(it) }.build(root, TextDomain.LITERALS)
}
return null
}
@@ -25,6 +25,7 @@ internal class PythonTextExtractor : TextExtractor() {
val elementType = PsiUtilCore.getElementType(root)
if (elementType in PyTokenTypes.STRING_NODES) {
val domain = if (elementType == PyTokenTypes.DOCSTRING) TextDomain.DOCUMENTATION else TextDomain.LITERALS
if (domain !in allowedDomains) return null
val stringContent = TextContentBuilder.FromPsi.removingIndents(" \t")
.removingLineSuffixes(" \t")
.withUnknown(this::isUnknownFragment)
@@ -35,7 +36,7 @@ internal class PythonTextExtractor : TextExtractor() {
return stringContent
}
if (root is PsiCommentImpl) {
if (root is PsiCommentImpl && TextDomain.COMMENTS in allowedDomains) {
val siblings = getNotSoDistantSimilarSiblings(root) { it is PsiCommentImpl }
return TextContent.joinWithWhitespace('\n', siblings.mapNotNull { TextContent.builder().build(it, TextDomain.COMMENTS) })
}