[grazie] IJPL-190645 Python: Perform text-level spellchecking in textual fragments

Merge-request: IJ-MR-178844
Merged-by: Ilia Permiashkin <ilia.permiashkin@jetbrains.com>

GitOrigin-RevId: e4e494dfdb80660e2651982941243de210cffbd8
This commit is contained in:
Ilia Permiashkin
2025-10-18 13:56:01 +00:00
committed by intellij-monorepo-bot
parent f150638f28
commit 3b26b8c52f
9 changed files with 90 additions and 55 deletions
@@ -133,8 +133,6 @@
<grazie.textChecker implementation="com.intellij.grazie.text.AsyncTreeRuleChecker$GrammarLowPriority" id="grammarLowPriority"/>
<grazie.textChecker implementation="com.intellij.grazie.text.AsyncTreeRuleChecker$Style" id="styleTreeRules"/>
<grazie.problemFilter language="" implementationClass="com.intellij.grazie.text.TreeRuleChecker$DocProblemFilter"/>
<intentionAction>
<language/>
<bundleName>messages.GrazieBundle</bundleName>
@@ -58,8 +58,6 @@ import java.net.URL;
import java.util.*;
import java.util.concurrent.atomic.AtomicReference;
import java.util.function.Function;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
import static com.intellij.grazie.text.GrazieProblem.getQuickFixText;
import static com.intellij.grazie.text.GrazieProblem.visualizeSpace;
@@ -606,29 +604,4 @@ public final class TreeRuleChecker {
return match.rule().shouldSuppressInCodeLikeFragments();
}
}
public static class DocProblemFilter extends ProblemFilter {
private static final Pattern PY_DOC_PARAM = Pattern.compile("[a-z0-9_]+\\s*:\\s+\\p{L}+( or \\p{L}+)*");
@Override
public boolean shouldIgnore(@NotNull TextProblem problem) {
TextContent text = problem.getText();
if (text.getDomain() == TextDomain.DOCUMENTATION) {
List<TextRange> ranges = problem.getHighlightRanges();
String psiClass = text.getCommonParent().getClass().getName();
//todo remove after https://youtrack.jetbrains.com/issue/PY-59061 is fixed
if (psiClass.equals("com.jetbrains.python.psi.impl.PyStringLiteralExpressionImpl")) {
Matcher matcher = PY_DOC_PARAM.matcher(text);
while (matcher.find()) {
if (ContainerUtil.exists(ranges, r -> r.intersects(matcher.start(), matcher.end()))) {
return true;
}
}
}
}
return false;
}
}
}
@@ -5,6 +5,10 @@ import com.intellij.grazie.text.RuleGroup
import com.intellij.grazie.text.TextContent
import com.intellij.grazie.text.TextProblem
import com.intellij.grazie.utils.ProblemFilterUtil
import com.jetbrains.python.psi.impl.PyStringLiteralExpressionImpl
import java.util.regex.Pattern
private val PY_DOC_PARAM = Pattern.compile("[a-z0-9_]+\\s*:\\s+\\p{L}+( or \\p{L}+)*")
private class PythonProblemFilter : ProblemFilter() {
override fun shouldIgnore(problem: TextProblem): Boolean {
@@ -12,6 +16,21 @@ private class PythonProblemFilter : ProblemFilter() {
return domain == TextContent.TextDomain.DOCUMENTATION &&
(ProblemFilterUtil.isUndecoratedSingleSentenceIssue(problem) ||
ProblemFilterUtil.isInitialCasingIssue(problem) ||
problem.fitsGroup(RuleGroup(RuleGroup.UNDECORATED_SENTENCE_SEPARATION)))
problem.fitsGroup(RuleGroup(RuleGroup.UNDECORATED_SENTENCE_SEPARATION)) ||
fitsPyDocParam(problem))
}
private fun fitsPyDocParam(problem: TextProblem): Boolean {
val text = problem.text
//todo remove after https://youtrack.jetbrains.com/issue/PY-59061 is fixed
if (text.domain == TextContent.TextDomain.DOCUMENTATION && text.getCommonParent().getParent() is PyStringLiteralExpressionImpl) {
val matcher = PY_DOC_PARAM.matcher(text)
while (matcher.find()) {
if (problem.highlightRanges.any { it.intersects(matcher.start(), matcher.end()) }) {
return true
}
}
}
return false
}
}
@@ -7,49 +7,75 @@ import com.intellij.grazie.text.TextContentBuilder
import com.intellij.grazie.text.TextExtractor
import com.intellij.grazie.utils.Text
import com.intellij.grazie.utils.getNotSoDistantSimilarSiblings
import com.intellij.openapi.util.TextRange
import com.intellij.psi.PsiElement
import com.intellij.psi.impl.source.tree.LeafPsiElement
import com.intellij.psi.impl.source.tree.PsiCommentImpl
import com.intellij.psi.util.PsiUtilCore
import com.jetbrains.python.PyStringFormatParser
import com.jetbrains.python.PyStringFormatParser.ConstantChunk
import com.jetbrains.python.PyTokenTypes
import com.jetbrains.python.PyTokenTypes.FSTRING_TEXT
import com.jetbrains.python.documentation.docstrings.SphinxDocString
import com.jetbrains.python.psi.PyBinaryExpression
import com.jetbrains.python.psi.PyFormattedStringElement
import com.jetbrains.python.psi.PyStringLiteralExpression
import com.jetbrains.python.psi.impl.PyStringLiteralDecoder
import java.util.regex.Pattern
import java.util.regex.Pattern.quote
private val KNOWN_DOCSTRING_TAGS_PATTERN = SphinxDocString.ALL_TAGS.joinToString("|", transform = Pattern::quote, prefix = "(", postfix = ")")
private val DOCSTRING_DIRECTIVE_PATTERN = "^$KNOWN_DOCSTRING_TAGS_PATTERN[^\n:]*: *".toPattern(Pattern.MULTILINE)
internal class PythonTextExtractor : TextExtractor() {
override fun buildTextContent(root: PsiElement, allowedDomains: MutableSet<TextDomain>): TextContent? {
val elementType = PsiUtilCore.getElementType(root)
if (elementType in PyTokenTypes.STRING_NODES) {
val domain = if (elementType == PyTokenTypes.DOCSTRING) TextDomain.DOCUMENTATION else TextDomain.LITERALS
if (domain !in allowedDomains) return null
val stringContent = TextContentBuilder.FromPsi.removingIndents(" \t")
.removingLineSuffixes(" \t")
.withUnknown(this::isUnknownFragment)
.build(root.parent, domain)
if (stringContent != null && domain == TextDomain.DOCUMENTATION && TextDomain.DOCUMENTATION in allowedDomains) {
return stringContent.excludeDocstringTags()
override fun buildTextContents(root: PsiElement, allowedDomains: Set<TextDomain>): List<TextContent> {
if (root is PyStringLiteralExpression) {
val texts = mutableListOf<TextContent>()
val parent = root.parent
if (parent is PyBinaryExpression && root === parent.leftExpression && parent.operator === PyTokenTypes.PERC) {
PyStringFormatParser.parsePercentFormat(root.stringValue).forEach { chunk ->
if (chunk is ConstantChunk) {
val startIndex = root.valueOffsetToTextOffset(chunk.startIndex)
val endIndex = root.valueOffsetToTextOffset(chunk.endIndex)
TextContentBuilder.FromPsi.build(root, TextDomain.LITERALS, TextRange(startIndex, endIndex))?.let { texts.add(it) }
}
}
return texts
}
return stringContent
root.stringElements.forEach { element ->
val ranges = if (element.isFormatted) (element as PyFormattedStringElement).literalPartRanges else listOf(element.contentRange)
val decoder = PyStringLiteralDecoder(element)
val containsEscapes = element.textContains('\\')
ranges.forEach { range ->
val escapeAwareRanges = if (element.isRaw || !containsEscapes) listOf(range) else decoder.decodeRange(range).map { it.first }
escapeAwareRanges.forEach { escapeAwareRange ->
val domain = getDomain(element)
TextContentBuilder.FromPsi
.removingIndents(" \t")
.removingLineSuffixes(" \t")
.build(element, domain, escapeAwareRange)?.let { text ->
if (domain == TextDomain.DOCUMENTATION) text.excludeDocstringTags()?.let { texts.add(it) } else texts.add(text)
}
}
}
}
return texts
}
if (root is PsiCommentImpl && TextDomain.COMMENTS in allowedDomains) {
val siblings = getNotSoDistantSimilarSiblings(root) { it is PsiCommentImpl }
return TextContent.joinWithWhitespace('\n', siblings.mapNotNull { TextContent.builder().build(it, TextDomain.COMMENTS) })
val text = TextContent.joinWithWhitespace(
'\n',
siblings.mapNotNull { TextContent.builder().build(it, TextDomain.COMMENTS) }
) ?: return emptyList()
return listOf(text)
}
return null
return emptyList()
}
private fun isUnknownFragment(element: PsiElement): Boolean {
if (element.parent is PyFormattedStringElement) {
return element !is LeafPsiElement || element.elementType != FSTRING_TEXT
}
return false
private fun getDomain(element: PsiElement): TextDomain {
val elementType = PsiUtilCore.getElementType(element)
return if (elementType == PyTokenTypes.DOCSTRING) TextDomain.DOCUMENTATION else TextDomain.LITERALS
}
}
@@ -4,6 +4,7 @@ package com.jetbrains.python.spellchecker;
import com.intellij.lang.injection.InjectedLanguageManager;
import com.intellij.openapi.project.DumbAware;
import com.intellij.openapi.util.TextRange;
import com.intellij.openapi.util.registry.Registry;
import com.intellij.psi.PsiElement;
import com.intellij.spellchecker.inspections.PlainTextSplitter;
import com.intellij.spellchecker.inspections.Splitter;
@@ -20,6 +21,7 @@ import com.jetbrains.python.psi.PyStringLiteralExpression;
import com.jetbrains.python.psi.impl.PyStringLiteralDecoder;
import org.jetbrains.annotations.NotNull;
import java.util.ArrayList;
import java.util.Collections;
import java.util.List;
@@ -73,12 +75,17 @@ public final class PythonSpellcheckerStrategy extends SpellcheckingStrategy impl
}
}
@Override
public boolean useTextLevelSpellchecking() {
return Registry.is("spellchecker.grazie.enabled", false);
}
private final StringLiteralTokenizer myStringLiteralTokenizer = new StringLiteralTokenizer();
private final FormatStringTokenizer myFormatStringTokenizer = new FormatStringTokenizer();
@Override
public @NotNull Tokenizer getTokenizer(PsiElement element) {
if (element instanceof PyStringLiteralExpression) {
if (element instanceof PyStringLiteralExpression && !useTextLevelSpellchecking()) {
final InjectedLanguageManager injectionManager = InjectedLanguageManager.getInstance(element.getProject());
if (element.getTextLength() >= 2 && injectionManager.getInjectedPsiFiles(element) != null) {
return EMPTY_TOKENIZER;
+7
View File
@@ -0,0 +1,7 @@
world = "World!"
some_text = "lots of cats"
fstring1 = f"Hello, ${world} There are $some_text. And here are some correct English words to make the language detector work. And a <TYPO descr="Typo: In word 'typpo'">typpo</TYPO>"
fstring1 = f"The <GRAMMAR_ERROR descr="COMMA_WHICH">name which</GRAMMAR_ERROR> group and some other English text. And a <TYPO descr="Typo: In word 'typpo'">typpo</TYPO>"
@@ -1 +1 @@
('\ncorrect' r'\<TYPO descr="Typo: In word 'ncorrect'">ncorrect</TYPO>')
('\ncorrect' r'\ncorrect')
@@ -1,4 +1,5 @@
'''
\n \t \\
@<TYPO descr="Typo: In word 'xyzzymethod'">xyzzymethod</TYPO>
@xyzzymethod
<TYPO descr="Typo: In word 'typopo'">typopo</TYPO>
_fields = %(field_names)r'''% locals()
@@ -10,6 +10,10 @@ class PythonGrazieSupportTest : GrazieTestBase() {
override fun getBasePath() = "python/testData/grazie/"
override fun isCommunity() = true
fun `test f-strings`() {
runHighlightTestForFile("FStrings.py")
}
fun `test grammar check in constructs`() {
runHighlightTestForFile("Constructs.py")
}