[spelling] IDEA-160183 Do not spellcheck md5, sha1 and other values

GitOrigin-RevId: a0955a7a7b4c6a418eed9fac22428c7267bee01f
This commit is contained in:
Yuriy Artamonov
2023-01-16 14:17:56 +00:00
committed by intellij-monorepo-bot
parent a0760365f4
commit c329819328
9 changed files with 233 additions and 19 deletions
@@ -62,6 +62,22 @@ public class JsonSpellcheckerTest extends JsonTestCase {
myFixture.doHighlighting();
}
public void testHashesQuotedSpelling() {
myFixture.enableInspections(SpellCheckingInspection.class);
myFixture.configureByText("hashes.json", """
{
"typo": "<TYPO>hereistheerror</TYPO>",
"uuid": "f19c4bd2-4c11-4725-a613-06aaead4325e",
"md5": "79054025255fb1a26e4bc422adfebeed",
"sha1": "c3499c2729730aaff07efb8676a92dcb6f8a3f8f",
"sha256": "50d858e0985ecc7f60418aaf0cc5ab587f42c2570a884095a9e8ccacd0f6545c",
"jwt": "eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.eyJzdWIiOiIxMjM0NTY3ODkwIiwibmFtZSI6IkpvaG4gRG9lIiwiYWRtaW4iOnRydWV9.dyt0CoTl4WoVjAHI9Q_CwSKhl6d_9rhM3NrXuJttkao"
}
""".stripIndent());
myFixture.checkHighlighting(true, false, true);
}
@Override
protected String getTestDataPath() {
return super.getTestDataPath() + "/spellchecker";
@@ -17,10 +17,15 @@ package org.jetbrains.yaml;
import com.intellij.json.JsonSchemaSpellcheckerClient;
import com.intellij.lang.ASTNode;
import com.intellij.openapi.util.TextRange;
import com.intellij.psi.PsiElement;
import com.intellij.psi.impl.source.tree.LeafPsiElement;
import com.intellij.psi.tree.IElementType;
import com.intellij.spellchecker.inspections.PlainTextSplitter;
import com.intellij.spellchecker.tokenizer.SpellcheckingStrategy;
import com.intellij.spellchecker.tokenizer.TokenConsumer;
import com.intellij.spellchecker.tokenizer.Tokenizer;
import com.intellij.spellchecker.tokenizer.TokenizerBase;
import com.jetbrains.jsonSchema.ide.JsonSchemaService;
import com.jetbrains.jsonSchema.impl.JsonSchemaObject;
import org.jetbrains.annotations.NotNull;
@@ -28,10 +33,32 @@ import org.jetbrains.annotations.Nullable;
import org.jetbrains.yaml.psi.YAMLKeyValue;
import org.jetbrains.yaml.psi.YAMLScalar;
public class YAMLSpellcheckerStrategy extends SpellcheckingStrategy {
final class YAMLSpellcheckerStrategy extends SpellcheckingStrategy {
private final Tokenizer<PsiElement> myQuotedTextTokenizer = new TokenizerBase<>(PlainTextSplitter.getInstance()) {
@Override
public void tokenize(@NotNull PsiElement element, @NotNull TokenConsumer consumer) {
if (element instanceof LeafPsiElement) {
CharSequence chars = ((LeafPsiElement)element).getChars();
int length = chars.length();
if (length >= 2
&& (chars.charAt(0) == '\'' || chars.charAt(0) == '"')
&& (chars.charAt(length - 1) == '\'' || chars.charAt(length - 1) == '"')) {
// remove quotes from text analysis
int quotesLength = 1;
String text = chars.subSequence(quotesLength, length - quotesLength).toString();
consumer.consumeToken(element, text, false, quotesLength, TextRange.allOf(text), PlainTextSplitter.getInstance());
}
} else {
super.tokenize(element, consumer);
}
}
};
@NotNull
@Override
public Tokenizer getTokenizer(final PsiElement element) {
public Tokenizer<?> getTokenizer(final PsiElement element) {
final ASTNode node = element.getNode();
if (node != null){
final IElementType type = node.getElementType();
@@ -42,12 +69,17 @@ public class YAMLSpellcheckerStrategy extends SpellcheckingStrategy {
type == YAMLTokenTypes.SCALAR_STRING ||
type == YAMLTokenTypes.SCALAR_DSTRING ||
type == YAMLTokenTypes.COMMENT) {
if (new JsonSchemaSpellcheckerClientForYaml(element).matchesNameFromSchema()) {
return EMPTY_TOKENIZER;
}
else {
return TEXT_TOKENIZER;
if (type == YAMLTokenTypes.SCALAR_STRING ||
type == YAMLTokenTypes.SCALAR_DSTRING) {
return myQuotedTextTokenizer;
}
return TEXT_TOKENIZER;
}
}
return super.getTokenizer(element);
@@ -26,4 +26,34 @@ class YAMLSpellCheckerTest : BasePlatformTestCase() {
""".trimIndent())
myFixture.checkHighlighting(true, false, true)
}
fun testHashesQuotedSpelling() {
myFixture.enableInspections(SpellCheckingInspection::class.java)
myFixture.configureByText("hashes.yaml", """
data:
typo: '<TYPO>hereistheerror</TYPO>'
uuid: 'f19c4bd2-4c11-4725-a613-06aaead4325e'
md5: '79054025255fb1a26e4bc422adfebeed'
sha1: "c3499c2729730aaff07efb8676a92dcb6f8a3f8f"
sha256: "50d858e0985ecc7f60418aaf0cc5ab587f42c2570a884095a9e8ccacd0f6545c"
jwt: 'eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.eyJzdWIiOiIxMjM0NTY3ODkwIiwibmFtZSI6IkpvaG4gRG9lIiwiYWRtaW4iOnRydWV9.dyt0CoTl4WoVjAHI9Q_CwSKhl6d_9rhM3NrXuJttkao'
""".trimIndent())
myFixture.checkHighlighting(true, false, true)
}
fun testHashesUnquotedSpelling() {
myFixture.enableInspections(SpellCheckingInspection::class.java)
myFixture.configureByText("hashes.yaml", """
data:
typo: <TYPO>hereistheerror</TYPO>
uuid: f19c4bd2-4c11-4725-a613-06aaead4325e
md5: 79054025255fb1a26e4bc422adfebeed
sha1: c3499c2729730aaff07efb8676a92dcb6f8a3f8f
sha256: 50d858e0985ecc7f60418aaf0cc5ab587f42c2570a884095a9e8ccacd0f6545c
jwt: eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.eyJzdWIiOiIxMjM0NTY3ODkwIiwibmFtZSI6IkpvaG4gRG9lIiwiYWRtaW4iOnRydWV9.dyt0CoTl4WoVjAHI9Q_CwSKhl6d_9rhM3NrXuJttkao
""".trimIndent())
myFixture.checkHighlighting(true, false, true)
}
}
@@ -20,16 +20,16 @@ import com.intellij.openapi.util.TextRange;
import com.intellij.openapi.util.text.StringUtil;
import com.intellij.util.Consumer;
import org.jdom.Verifier;
import org.jetbrains.annotations.NonNls;
import org.jetbrains.annotations.NotNull;
import org.jetbrains.annotations.Nullable;
import java.util.Collections;
import java.util.List;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
import static com.intellij.util.io.URLUtil.URL_PATTERN;
import static java.util.Collections.emptyList;
import static java.util.Collections.singletonList;
public class PlainTextSplitter extends BaseSplitter {
private static final PlainTextSplitter INSTANCE = new PlainTextSplitter();
@@ -38,17 +38,38 @@ public class PlainTextSplitter extends BaseSplitter {
return INSTANCE;
}
@NonNls
private static final
Pattern SPLIT_PATTERN = Pattern.compile("(\\s|\b)");
@NonNls
private static final Pattern MAIL =
Pattern.compile("([\\p{L}0-9\\.\\-\\_\\+]+@([\\p{L}0-9\\-\\_]+(\\.)?)+(com|net|[a-z]{2})?)");
@NonNls
private static final int UUID_V4_HEX_STRING_LENGTH = 36;
private static final Pattern UUID_PATTERN = Pattern.compile("[a-fA-F0-9]{8}(-[a-fA-F0-9]{4}){3}-[a-fA-F0-9]{12}");
private static final int MD5_HEX_LENGTH = 32;
private static final int SHA1_HEX_LENGTH = 40;
private static final int SHA256_HEX_LENGTH = 64;
private static final int SHA512_HEX_LENGTH = 128;
private static final String HEX_SYMBOLS = "[0-9A-Fa-f]";
private static final Pattern MD5_HEX_PATTERN = Pattern.compile(HEX_SYMBOLS + "{" + MD5_HEX_LENGTH + "}");
private static final Pattern SHA1_HEX_PATTERN = Pattern.compile(HEX_SYMBOLS + "{" + SHA1_HEX_LENGTH + "}");
private static final Pattern SHA256_HEX_PATTERN = Pattern.compile(HEX_SYMBOLS + "{" + SHA256_HEX_LENGTH + "}");
private static final Pattern SHA512_HEX_PATTERN = Pattern.compile(HEX_SYMBOLS + "{" + SHA512_HEX_LENGTH + "}");
private static final int SHA384_BASE64_LENGTH = 64;
private static final String SHA384_PREFIX = "sha384-";
private final static Pattern SHA384_PREFIXED_VALUE_PATTERN = Pattern.compile("sha384-[A-Za-z0-9+=/]{" + SHA384_BASE64_LENGTH + "}");
private static final int SHA512_BASE64_LENGTH = 88;
private static final String SHA512_PREFIX = "sha512-";
private final static Pattern SHA512_PREFIXED_VALUE_PATTERN = Pattern.compile("sha512-[A-Za-z0-9+=/]{" + SHA512_BASE64_LENGTH + "}");
private final static String JWT_COMMON_PREFIX = "eyJhbGci"; // Base64 of `{"alg":` in JWT header
private final static Pattern JWT_PATTERN = Pattern.compile("[A-Za-z0-9+=/_\\-.]+");
@Override
public void split(@Nullable String text, @NotNull TextRange range, Consumer<TextRange> consumer) {
if (StringUtil.isEmpty(text)) {
@@ -68,9 +89,11 @@ public class PlainTextSplitter extends BaseSplitter {
while (true) {
checkCancelled();
List<TextRange> toCheck;
TextRange wRange;
String word;
if (matcher.find()) {
TextRange found = matcherRange(range, matcher);
till = found.getStartOffset();
@@ -86,21 +109,39 @@ public class PlainTextSplitter extends BaseSplitter {
wRange = new TextRange(from, range.getEndOffset());
word = wRange.substring(text);
}
int wordLength = word.length();
if (word.contains("@")) {
toCheck = excludeByPattern(text, wRange, MAIL, 0);
}
else if (word.contains("://")) {
toCheck = excludeByPattern(text, wRange, URL_PATTERN, 0);
}
else if (word.contains("-")) {
toCheck = excludeByPattern(text, wRange, UUID_PATTERN, 0);
else if (word.startsWith(JWT_COMMON_PREFIX) && JWT_PATTERN.matcher(word).matches()) {
toCheck = emptyList();
}
else if (wordLength == MD5_HEX_LENGTH && MD5_HEX_PATTERN.matcher(word).matches() ||
wordLength == SHA1_HEX_LENGTH && SHA1_HEX_PATTERN.matcher(word).matches() ||
wordLength == SHA256_HEX_LENGTH && SHA256_HEX_PATTERN.matcher(word).matches() ||
wordLength == SHA512_HEX_LENGTH && SHA512_HEX_PATTERN.matcher(word).matches()) {
toCheck = emptyList();
}
else if (wordLength == UUID_V4_HEX_STRING_LENGTH && UUID_PATTERN.matcher(word).matches()) {
toCheck = emptyList();
}
else if (isHashPrefixed(word, SHA384_PREFIX, SHA384_BASE64_LENGTH) && SHA384_PREFIXED_VALUE_PATTERN.matcher(word).matches()
|| isHashPrefixed(word, SHA512_PREFIX, SHA512_BASE64_LENGTH) && SHA512_PREFIXED_VALUE_PATTERN.matcher(word).matches()) {
toCheck = emptyList(); // various integrity
}
else {
toCheck = Collections.singletonList(wRange);
toCheck = singletonList(wRange);
}
for (TextRange r : toCheck) {
ws.split(text, r, consumer);
}
if (matcher.hitEnd()) break;
}
}
@@ -112,4 +153,9 @@ public class PlainTextSplitter extends BaseSplitter {
protected Splitter getTextSplitter() {
return TextSplitter.getInstance();
}
private static boolean isHashPrefixed(String text, String hashPrefix, int expectedHashSize) {
return text.length() == expectedHashSize + hashPrefix.length()
&& text.startsWith(hashPrefix);
}
}
@@ -2,18 +2,29 @@
package com.intellij.spellchecker.tokenizer;
import com.intellij.openapi.util.TextRange;
import com.intellij.psi.ElementManipulators;
import com.intellij.psi.PsiComment;
import com.intellij.psi.PsiElement;
import com.intellij.psi.PsiLanguageInjectionHost;
import com.intellij.spellchecker.inspections.Splitter;
public abstract class TokenConsumer {
public void consumeToken(PsiElement element, Splitter splitter) {
consumeToken(element, false, splitter);
}
public void consumeToken(PsiElement element, boolean useRename, Splitter splitter) {
String text = element.getText();
consumeToken(element, text, useRename, 0, TextRange.allOf(text), splitter);
if (element instanceof PsiLanguageInjectionHost && !(element instanceof PsiComment)) {
// remove quotes from text analysis
TextRange range = ElementManipulators.getValueTextRange(element);
if (!range.isEmpty()) {
String text = ElementManipulators.getValueText(element);
consumeToken(element, text, useRename, range.getStartOffset(), TextRange.allOf(text), splitter);
}
} else {
String text = element.getText();
consumeToken(element, text, useRename, 0, TextRange.allOf(text), splitter);
}
}
/**
@@ -0,0 +1,40 @@
<html lang="en">
<head>
<title>Toolbox Enterprise Mock Auth Login</title>
<link rel="preconnect" href="https://fonts.gstatic.com"/>
<link href="https://fonts.googleapis.com/css2?family=JetBrains+Mono&amp;display=swap" rel="stylesheet"/>
<script src="https://cdnjs.cloudflare.com/ajax/libs/jquery/3.6.0/jquery.min.js"
integrity="sha512-894YE6QWD5I59HgZOGReFYm4dnWc1Qt5NtvYSaNcOP+u1T9qYdvdihz0PPSiiqn/+/3e7Jo4EaG7TubfWGUrMQ=="
crossorigin="anonymous"
referrerpolicy="no-referrer"></script>
<link rel="stylesheet" href="https://stackpath.bootstrapcdn.com/bootstrap/3.4.1/css/bootstrap.min.css"
integrity="sha384-HSMxcRTRxnN+Bdg0JdbxYKrThecOKuH5zCYotlSAcp1+c8xmyTe9GYg1l9a69psu" crossorigin="anonymous">
<link rel="stylesheet" href="https://stackpath.bootstrapcdn.com/bootstrap/3.4.1/css/bootstrap-theme.min.css"
integrity="sha384-6pzBo3FDv/PJ8r2KRkGHifhEocL+1X2rVCTTkUfGk7/0pbek5mMa1upzvWbrUbOZ" crossorigin="anonymous">
<style>
html {
padding: 5em;
}
body {
font-family: 'JetBrains Mono', monospace;
}
a, a:visited {
color: #167dff;
}
tr {
padding: 5em;
}
button {
padding: 1em;
width: 20ex;
border-radius: 5px;
}
</style>
</head>
<body>
<a title="<TYPO>linkwitherror</TYPO>"><TYPO descr="Typo: In word 'Linkwitherror'">Linkwitherror</TYPO></a>
</body>
@@ -40,4 +40,8 @@ public class XmlWithMistakesInspectionTest extends SpellcheckerInspectionTestCas
// "evenodd" is correct word in SVG, because it is known enumeration option in SVG
doTest("enumerations.svg");
}
public void testLinkIntegrity() {
doTest("htmlIntegrity.html");
}
}
@@ -422,6 +422,30 @@ public class SplitterTest {
assertEquals(0, words.size());
}
@Test
public void testMd5InsideText() {
String text = "asdasd 79054025255fb1a26e4bc422adfebeed asdasd";
correctListToCheck(PlainTextSplitter.getInstance(), text, "asdasd", "asdasd");
}
@Test
public void testSha1InsideText() {
String text = "asdasd c3499c2729730aaff07efb8676a92dcb6f8a3f8f asdasd";
correctListToCheck(PlainTextSplitter.getInstance(), text, "asdasd", "asdasd");
}
@Test
public void testSha256InsideText() {
String text = "asdasd 50d858e0985ecc7f60418aaf0cc5ab587f42c2570a884095a9e8ccacd0f6545c asdasd";
correctListToCheck(PlainTextSplitter.getInstance(), text, "asdasd", "asdasd");
}
@Test
public void testJwtInsideText() {
String text = "asdasd eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.eyJzdWIiOiIxMjM0NTY3ODkwIiwibmFtZSI6IkpvaG4gRG9lIiwiYWRtaW4iOnRydWV9.dyt0CoTl4WoVjAHI9Q_CwSKhl6d_9rhM3NrXuJttkao asdasd";
correctListToCheck(PlainTextSplitter.getInstance(), text, "asdasd", "asdasd");
}
@NotNull
private static List<String> wordsToCheck(Splitter splitter, final String text) {
final List<String> words = new ArrayList<>();
@@ -3,10 +3,7 @@ package com.intellij.spellchecker.xml;
import com.intellij.codeInspection.SuppressQuickFix;
import com.intellij.openapi.util.TextRange;
import com.intellij.openapi.util.text.StringUtil;
import com.intellij.psi.PsiElement;
import com.intellij.psi.PsiFile;
import com.intellij.psi.PsiReference;
import com.intellij.psi.XmlElementVisitor;
import com.intellij.psi.*;
import com.intellij.psi.impl.source.tree.LeafPsiElement;
import com.intellij.psi.templateLanguages.TemplateLanguage;
import com.intellij.psi.tree.IElementType;
@@ -27,6 +24,9 @@ import org.jetbrains.annotations.Nullable;
import java.util.List;
import static java.util.Collections.emptyList;
import static java.util.Collections.singletonList;
public class XmlSpellcheckingStrategy extends SuppressibleSpellcheckingStrategy {
private final Tokenizer<? extends PsiElement> myXmlTextTokenizer = createTextTokenizer();
@@ -152,6 +152,16 @@ public class XmlSpellcheckingStrategy extends SuppressibleSpellcheckingStrategy
|| XmlTokenType.WHITESPACES.contains(tokenType);
}
@Override
protected @NotNull List<@NotNull SpellcheckRange> getSpellcheckRanges(@NotNull XmlAttributeValue element) {
TextRange range = ElementManipulators.getValueTextRange(element);
if (range.isEmpty()) return emptyList();
String text = ElementManipulators.getValueText(element);
return singletonList(new SpellcheckRange(text, false, range.getStartOffset(), TextRange.allOf(text)));
}
@Override
public void tokenize(@NotNull XmlAttributeValue element, TokenConsumer consumer) {
PsiReference[] references = element.getReferences();
@@ -169,6 +179,7 @@ public class XmlSpellcheckingStrategy extends SuppressibleSpellcheckingStrategy
if (valueTextTrimmed.startsWith("#") && valueTextTrimmed.length() <= 9 && isHexString(valueTextTrimmed.substring(1))) {
return;
}
super.tokenize(element, consumer);
}