mirror of
https://gitflic.ru/project/openide/openide.git
synced 2026-09-27 10:03:11 +07:00
[grazie] IJPL-190645 Python: Perform text-level spellchecking in textual fragments
Merge-request: IJ-MR-178844 Merged-by: Ilia Permiashkin <ilia.permiashkin@jetbrains.com> GitOrigin-RevId: e4e494dfdb80660e2651982941243de210cffbd8
This commit is contained in:
committed by
intellij-monorepo-bot
parent
f150638f28
commit
3b26b8c52f
@@ -133,8 +133,6 @@
|
||||
<grazie.textChecker implementation="com.intellij.grazie.text.AsyncTreeRuleChecker$GrammarLowPriority" id="grammarLowPriority"/>
|
||||
<grazie.textChecker implementation="com.intellij.grazie.text.AsyncTreeRuleChecker$Style" id="styleTreeRules"/>
|
||||
|
||||
<grazie.problemFilter language="" implementationClass="com.intellij.grazie.text.TreeRuleChecker$DocProblemFilter"/>
|
||||
|
||||
<intentionAction>
|
||||
<language/>
|
||||
<bundleName>messages.GrazieBundle</bundleName>
|
||||
|
||||
@@ -58,8 +58,6 @@ import java.net.URL;
|
||||
import java.util.*;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Function;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import static com.intellij.grazie.text.GrazieProblem.getQuickFixText;
|
||||
import static com.intellij.grazie.text.GrazieProblem.visualizeSpace;
|
||||
@@ -606,29 +604,4 @@ public final class TreeRuleChecker {
|
||||
return match.rule().shouldSuppressInCodeLikeFragments();
|
||||
}
|
||||
}
|
||||
|
||||
public static class DocProblemFilter extends ProblemFilter {
|
||||
private static final Pattern PY_DOC_PARAM = Pattern.compile("[a-z0-9_]+\\s*:\\s+\\p{L}+( or \\p{L}+)*");
|
||||
|
||||
@Override
|
||||
public boolean shouldIgnore(@NotNull TextProblem problem) {
|
||||
TextContent text = problem.getText();
|
||||
if (text.getDomain() == TextDomain.DOCUMENTATION) {
|
||||
List<TextRange> ranges = problem.getHighlightRanges();
|
||||
String psiClass = text.getCommonParent().getClass().getName();
|
||||
|
||||
//todo remove after https://youtrack.jetbrains.com/issue/PY-59061 is fixed
|
||||
if (psiClass.equals("com.jetbrains.python.psi.impl.PyStringLiteralExpressionImpl")) {
|
||||
Matcher matcher = PY_DOC_PARAM.matcher(text);
|
||||
while (matcher.find()) {
|
||||
if (ContainerUtil.exists(ranges, r -> r.intersects(matcher.start(), matcher.end()))) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,6 +5,10 @@ import com.intellij.grazie.text.RuleGroup
|
||||
import com.intellij.grazie.text.TextContent
|
||||
import com.intellij.grazie.text.TextProblem
|
||||
import com.intellij.grazie.utils.ProblemFilterUtil
|
||||
import com.jetbrains.python.psi.impl.PyStringLiteralExpressionImpl
|
||||
import java.util.regex.Pattern
|
||||
|
||||
private val PY_DOC_PARAM = Pattern.compile("[a-z0-9_]+\\s*:\\s+\\p{L}+( or \\p{L}+)*")
|
||||
|
||||
private class PythonProblemFilter : ProblemFilter() {
|
||||
override fun shouldIgnore(problem: TextProblem): Boolean {
|
||||
@@ -12,6 +16,21 @@ private class PythonProblemFilter : ProblemFilter() {
|
||||
return domain == TextContent.TextDomain.DOCUMENTATION &&
|
||||
(ProblemFilterUtil.isUndecoratedSingleSentenceIssue(problem) ||
|
||||
ProblemFilterUtil.isInitialCasingIssue(problem) ||
|
||||
problem.fitsGroup(RuleGroup(RuleGroup.UNDECORATED_SENTENCE_SEPARATION)))
|
||||
problem.fitsGroup(RuleGroup(RuleGroup.UNDECORATED_SENTENCE_SEPARATION)) ||
|
||||
fitsPyDocParam(problem))
|
||||
}
|
||||
|
||||
private fun fitsPyDocParam(problem: TextProblem): Boolean {
|
||||
val text = problem.text
|
||||
//todo remove after https://youtrack.jetbrains.com/issue/PY-59061 is fixed
|
||||
if (text.domain == TextContent.TextDomain.DOCUMENTATION && text.getCommonParent().getParent() is PyStringLiteralExpressionImpl) {
|
||||
val matcher = PY_DOC_PARAM.matcher(text)
|
||||
while (matcher.find()) {
|
||||
if (problem.highlightRanges.any { it.intersects(matcher.start(), matcher.end()) }) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
}
|
||||
@@ -7,49 +7,75 @@ import com.intellij.grazie.text.TextContentBuilder
|
||||
import com.intellij.grazie.text.TextExtractor
|
||||
import com.intellij.grazie.utils.Text
|
||||
import com.intellij.grazie.utils.getNotSoDistantSimilarSiblings
|
||||
import com.intellij.openapi.util.TextRange
|
||||
import com.intellij.psi.PsiElement
|
||||
import com.intellij.psi.impl.source.tree.LeafPsiElement
|
||||
import com.intellij.psi.impl.source.tree.PsiCommentImpl
|
||||
import com.intellij.psi.util.PsiUtilCore
|
||||
import com.jetbrains.python.PyStringFormatParser
|
||||
import com.jetbrains.python.PyStringFormatParser.ConstantChunk
|
||||
import com.jetbrains.python.PyTokenTypes
|
||||
import com.jetbrains.python.PyTokenTypes.FSTRING_TEXT
|
||||
import com.jetbrains.python.documentation.docstrings.SphinxDocString
|
||||
import com.jetbrains.python.psi.PyBinaryExpression
|
||||
import com.jetbrains.python.psi.PyFormattedStringElement
|
||||
import com.jetbrains.python.psi.PyStringLiteralExpression
|
||||
import com.jetbrains.python.psi.impl.PyStringLiteralDecoder
|
||||
import java.util.regex.Pattern
|
||||
import java.util.regex.Pattern.quote
|
||||
|
||||
private val KNOWN_DOCSTRING_TAGS_PATTERN = SphinxDocString.ALL_TAGS.joinToString("|", transform = Pattern::quote, prefix = "(", postfix = ")")
|
||||
private val DOCSTRING_DIRECTIVE_PATTERN = "^$KNOWN_DOCSTRING_TAGS_PATTERN[^\n:]*: *".toPattern(Pattern.MULTILINE)
|
||||
|
||||
internal class PythonTextExtractor : TextExtractor() {
|
||||
override fun buildTextContent(root: PsiElement, allowedDomains: MutableSet<TextDomain>): TextContent? {
|
||||
val elementType = PsiUtilCore.getElementType(root)
|
||||
if (elementType in PyTokenTypes.STRING_NODES) {
|
||||
val domain = if (elementType == PyTokenTypes.DOCSTRING) TextDomain.DOCUMENTATION else TextDomain.LITERALS
|
||||
if (domain !in allowedDomains) return null
|
||||
val stringContent = TextContentBuilder.FromPsi.removingIndents(" \t")
|
||||
.removingLineSuffixes(" \t")
|
||||
.withUnknown(this::isUnknownFragment)
|
||||
.build(root.parent, domain)
|
||||
if (stringContent != null && domain == TextDomain.DOCUMENTATION && TextDomain.DOCUMENTATION in allowedDomains) {
|
||||
return stringContent.excludeDocstringTags()
|
||||
override fun buildTextContents(root: PsiElement, allowedDomains: Set<TextDomain>): List<TextContent> {
|
||||
if (root is PyStringLiteralExpression) {
|
||||
val texts = mutableListOf<TextContent>()
|
||||
val parent = root.parent
|
||||
if (parent is PyBinaryExpression && root === parent.leftExpression && parent.operator === PyTokenTypes.PERC) {
|
||||
PyStringFormatParser.parsePercentFormat(root.stringValue).forEach { chunk ->
|
||||
if (chunk is ConstantChunk) {
|
||||
val startIndex = root.valueOffsetToTextOffset(chunk.startIndex)
|
||||
val endIndex = root.valueOffsetToTextOffset(chunk.endIndex)
|
||||
TextContentBuilder.FromPsi.build(root, TextDomain.LITERALS, TextRange(startIndex, endIndex))?.let { texts.add(it) }
|
||||
}
|
||||
}
|
||||
return texts
|
||||
}
|
||||
return stringContent
|
||||
|
||||
root.stringElements.forEach { element ->
|
||||
val ranges = if (element.isFormatted) (element as PyFormattedStringElement).literalPartRanges else listOf(element.contentRange)
|
||||
val decoder = PyStringLiteralDecoder(element)
|
||||
val containsEscapes = element.textContains('\\')
|
||||
ranges.forEach { range ->
|
||||
val escapeAwareRanges = if (element.isRaw || !containsEscapes) listOf(range) else decoder.decodeRange(range).map { it.first }
|
||||
escapeAwareRanges.forEach { escapeAwareRange ->
|
||||
val domain = getDomain(element)
|
||||
TextContentBuilder.FromPsi
|
||||
.removingIndents(" \t")
|
||||
.removingLineSuffixes(" \t")
|
||||
.build(element, domain, escapeAwareRange)?.let { text ->
|
||||
if (domain == TextDomain.DOCUMENTATION) text.excludeDocstringTags()?.let { texts.add(it) } else texts.add(text)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return texts
|
||||
}
|
||||
|
||||
if (root is PsiCommentImpl && TextDomain.COMMENTS in allowedDomains) {
|
||||
val siblings = getNotSoDistantSimilarSiblings(root) { it is PsiCommentImpl }
|
||||
return TextContent.joinWithWhitespace('\n', siblings.mapNotNull { TextContent.builder().build(it, TextDomain.COMMENTS) })
|
||||
val text = TextContent.joinWithWhitespace(
|
||||
'\n',
|
||||
siblings.mapNotNull { TextContent.builder().build(it, TextDomain.COMMENTS) }
|
||||
) ?: return emptyList()
|
||||
return listOf(text)
|
||||
}
|
||||
|
||||
return null
|
||||
return emptyList()
|
||||
}
|
||||
|
||||
private fun isUnknownFragment(element: PsiElement): Boolean {
|
||||
if (element.parent is PyFormattedStringElement) {
|
||||
return element !is LeafPsiElement || element.elementType != FSTRING_TEXT
|
||||
}
|
||||
|
||||
return false
|
||||
private fun getDomain(element: PsiElement): TextDomain {
|
||||
val elementType = PsiUtilCore.getElementType(element)
|
||||
return if (elementType == PyTokenTypes.DOCSTRING) TextDomain.DOCUMENTATION else TextDomain.LITERALS
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -4,6 +4,7 @@ package com.jetbrains.python.spellchecker;
|
||||
import com.intellij.lang.injection.InjectedLanguageManager;
|
||||
import com.intellij.openapi.project.DumbAware;
|
||||
import com.intellij.openapi.util.TextRange;
|
||||
import com.intellij.openapi.util.registry.Registry;
|
||||
import com.intellij.psi.PsiElement;
|
||||
import com.intellij.spellchecker.inspections.PlainTextSplitter;
|
||||
import com.intellij.spellchecker.inspections.Splitter;
|
||||
@@ -20,6 +21,7 @@ import com.jetbrains.python.psi.PyStringLiteralExpression;
|
||||
import com.jetbrains.python.psi.impl.PyStringLiteralDecoder;
|
||||
import org.jetbrains.annotations.NotNull;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.List;
|
||||
|
||||
@@ -73,12 +75,17 @@ public final class PythonSpellcheckerStrategy extends SpellcheckingStrategy impl
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean useTextLevelSpellchecking() {
|
||||
return Registry.is("spellchecker.grazie.enabled", false);
|
||||
}
|
||||
|
||||
private final StringLiteralTokenizer myStringLiteralTokenizer = new StringLiteralTokenizer();
|
||||
private final FormatStringTokenizer myFormatStringTokenizer = new FormatStringTokenizer();
|
||||
|
||||
@Override
|
||||
public @NotNull Tokenizer getTokenizer(PsiElement element) {
|
||||
if (element instanceof PyStringLiteralExpression) {
|
||||
if (element instanceof PyStringLiteralExpression && !useTextLevelSpellchecking()) {
|
||||
final InjectedLanguageManager injectionManager = InjectedLanguageManager.getInstance(element.getProject());
|
||||
if (element.getTextLength() >= 2 && injectionManager.getInjectedPsiFiles(element) != null) {
|
||||
return EMPTY_TOKENIZER;
|
||||
|
||||
@@ -0,0 +1,7 @@
|
||||
world = "World!"
|
||||
some_text = "lots of cats"
|
||||
|
||||
fstring1 = f"Hello, ${world} There are $some_text. And here are some correct English words to make the language detector work. And a <TYPO descr="Typo: In word 'typpo'">typpo</TYPO>"
|
||||
|
||||
fstring1 = f"The <GRAMMAR_ERROR descr="COMMA_WHICH">name which</GRAMMAR_ERROR> group and some other English text. And a <TYPO descr="Typo: In word 'typpo'">typpo</TYPO>"
|
||||
|
||||
@@ -1 +1 @@
|
||||
('\ncorrect' r'\<TYPO descr="Typo: In word 'ncorrect'">ncorrect</TYPO>')
|
||||
('\ncorrect' r'\ncorrect')
|
||||
@@ -1,4 +1,5 @@
|
||||
'''
|
||||
\n \t \\
|
||||
@<TYPO descr="Typo: In word 'xyzzymethod'">xyzzymethod</TYPO>
|
||||
@xyzzymethod
|
||||
<TYPO descr="Typo: In word 'typopo'">typopo</TYPO>
|
||||
_fields = %(field_names)r'''% locals()
|
||||
@@ -10,6 +10,10 @@ class PythonGrazieSupportTest : GrazieTestBase() {
|
||||
override fun getBasePath() = "python/testData/grazie/"
|
||||
override fun isCommunity() = true
|
||||
|
||||
fun `test f-strings`() {
|
||||
runHighlightTestForFile("FStrings.py")
|
||||
}
|
||||
|
||||
fun `test grammar check in constructs`() {
|
||||
runHighlightTestForFile("Constructs.py")
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user