mirror of
https://gitflic.ru/project/openide/openide.git
synced 2026-09-27 10:03:11 +07:00
Java text blocks: support octal and unicode escape sequences (IDEA-251084)
GitOrigin-RevId: 00a5992c83b0824c7630e5e5657a1f564b2c766f
This commit is contained in:
committed by
intellij-monorepo-bot
parent
d5ebd0fca2
commit
24ce3f8313
@@ -13,13 +13,13 @@ public class JavaHighlightingLexer extends LayeredLexer {
|
||||
public JavaHighlightingLexer(@NotNull LanguageLevel languageLevel) {
|
||||
super(JavaParserDefinition.createLexer(languageLevel));
|
||||
|
||||
registerSelfStoppingLayer(new StringLiteralLexer('\"', JavaTokenType.STRING_LITERAL, false, "s"),
|
||||
registerSelfStoppingLayer(new JavaStringLiteralLexer('\"', JavaTokenType.STRING_LITERAL, false, "s"),
|
||||
new IElementType[]{JavaTokenType.STRING_LITERAL}, IElementType.EMPTY_ARRAY);
|
||||
|
||||
registerSelfStoppingLayer(new StringLiteralLexer('\'', JavaTokenType.STRING_LITERAL),
|
||||
registerSelfStoppingLayer(new JavaStringLiteralLexer('\'', JavaTokenType.STRING_LITERAL),
|
||||
new IElementType[]{JavaTokenType.CHARACTER_LITERAL}, IElementType.EMPTY_ARRAY);
|
||||
|
||||
registerSelfStoppingLayer(new StringLiteralLexer(StringLiteralLexer.NO_QUOTE_CHAR, JavaTokenType.TEXT_BLOCK_LITERAL, true, "s"),
|
||||
registerSelfStoppingLayer(new JavaStringLiteralLexer(StringLiteralLexer.NO_QUOTE_CHAR, JavaTokenType.TEXT_BLOCK_LITERAL, true, "s"),
|
||||
new IElementType[]{JavaTokenType.TEXT_BLOCK_LITERAL}, IElementType.EMPTY_ARRAY);
|
||||
|
||||
LayeredLexer docLexer = new LayeredLexer(JavaParserDefinition.createDocLexer(languageLevel));
|
||||
|
||||
@@ -0,0 +1,67 @@
|
||||
// Copyright 2000-2020 JetBrains s.r.o. Use of this source code is governed by the Apache 2.0 license that can be found in the LICENSE file.
|
||||
package com.intellij.lexer;
|
||||
|
||||
import com.intellij.openapi.util.text.StringUtil;
|
||||
import com.intellij.psi.StringEscapesTokenTypes;
|
||||
import com.intellij.psi.tree.IElementType;
|
||||
import org.jetbrains.annotations.NotNull;
|
||||
|
||||
import java.util.stream.IntStream;
|
||||
|
||||
class JavaStringLiteralLexer extends StringLiteralLexer {
|
||||
|
||||
JavaStringLiteralLexer(char quoteChar, IElementType originalLiteralToken) {
|
||||
super(quoteChar, originalLiteralToken);
|
||||
}
|
||||
|
||||
JavaStringLiteralLexer(char quoteChar,
|
||||
IElementType originalLiteralToken,
|
||||
boolean canEscapeEolOrFramingSpaces,
|
||||
String additionalValidEscapes) {
|
||||
super(quoteChar, originalLiteralToken, canEscapeEolOrFramingSpaces, additionalValidEscapes);
|
||||
}
|
||||
|
||||
@Override
|
||||
protected @NotNull IElementType getUnicodeEscapeSequenceType() {
|
||||
int start = myStart + 2;
|
||||
while (start < myEnd && myBuffer.charAt(start) == 'u') start++;
|
||||
if (start + 3 >= myEnd) return StringEscapesTokenTypes.INVALID_UNICODE_ESCAPE_TOKEN;
|
||||
if (IntStream.range(start, start + 4).anyMatch(i -> !StringUtil.isHexDigit(myBuffer.charAt(i)))) {
|
||||
return StringEscapesTokenTypes.INVALID_UNICODE_ESCAPE_TOKEN;
|
||||
}
|
||||
return StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN;
|
||||
}
|
||||
|
||||
@Override
|
||||
protected int locateUnicodeEscapeSequence(int start, int i) {
|
||||
i++;
|
||||
while (i < myBufferEnd && myBuffer.charAt(i) == 'u') i++;
|
||||
int end = parseUnicodeDigits(i);
|
||||
if (end != i + 4) return end;
|
||||
int code = Integer.parseInt(myBuffer.subSequence(i, end).toString(), 16);
|
||||
i = end;
|
||||
// if escape sequence is not translated to backspace then continue from the next symbol
|
||||
if (code != '\\' || i >= myBufferEnd) return i;
|
||||
char c = myBuffer.charAt(i);
|
||||
if (StringUtil.isOctalDigit(c)) {
|
||||
if (i + 2 < myBufferEnd && StringUtil.isOctalDigit(myBuffer.charAt(i + 1)) && StringUtil.isOctalDigit(myBuffer.charAt(i + 1))) {
|
||||
return i + 3;
|
||||
}
|
||||
}
|
||||
else if (c == '\\' && i + 1 < myBufferEnd && myBuffer.charAt(i + 1) == 'u') {
|
||||
i++;
|
||||
while (i < myBufferEnd && myBuffer.charAt(i) == 'u') i++;
|
||||
return parseUnicodeDigits(i);
|
||||
}
|
||||
return i + 1;
|
||||
}
|
||||
|
||||
private int parseUnicodeDigits(int i) {
|
||||
int end = i + 4;
|
||||
for (; i < end; i++) {
|
||||
if (i == myBufferEnd) return i;
|
||||
if (!StringUtil.isHexDigit(myBuffer.charAt(i))) return i;
|
||||
}
|
||||
return end;
|
||||
}
|
||||
}
|
||||
@@ -11,6 +11,8 @@ import org.jetbrains.annotations.NonNls;
|
||||
import org.jetbrains.annotations.NotNull;
|
||||
import org.jetbrains.annotations.Nullable;
|
||||
|
||||
import java.util.function.Predicate;
|
||||
|
||||
public final class PsiLiteralUtil {
|
||||
@NonNls public static final String HEX_PREFIX = "0x";
|
||||
@NonNls public static final String BIN_PREFIX = "0b";
|
||||
@@ -475,7 +477,7 @@ public final class PsiLiteralUtil {
|
||||
for (int i = 0; i < lines.length; i++) {
|
||||
String line = lines[i];
|
||||
if (line.length() > 0) {
|
||||
sb.append(trimTrailingWhitespace(line.substring(prefix)));
|
||||
sb.append(trimTrailingWhitespaces(line.substring(prefix)));
|
||||
}
|
||||
if (i < lines.length - 1) {
|
||||
sb.append('\n');
|
||||
@@ -485,11 +487,62 @@ public final class PsiLiteralUtil {
|
||||
}
|
||||
|
||||
@NotNull
|
||||
private static String trimTrailingWhitespace(@NotNull String line) {
|
||||
int index = line.length() - 1;
|
||||
while (index >= 0 && Character.isWhitespace(line.charAt(index))) index--;
|
||||
if (index >= 0 && index < line.length() - 1 && line.charAt(index) == '\\') index++;
|
||||
return line.substring(0, index + 1);
|
||||
private static String trimTrailingWhitespaces(@NotNull String line) {
|
||||
int index = line.length();
|
||||
while (true) {
|
||||
int wsIndex = parseWhitespaceBackwards(line, index - 1);
|
||||
if (wsIndex == -1) break;
|
||||
index = wsIndex;
|
||||
}
|
||||
return line.substring(0, index);
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse whitespace (possibly escaped) starting from its last character.
|
||||
*
|
||||
* @return -1 if sequence is not a whitespace or whitespace is escaped
|
||||
*/
|
||||
private static int parseWhitespaceBackwards(@NotNull String s, int index) {
|
||||
if (index < 0) return -1;
|
||||
if (Character.isWhitespace(s.charAt(index))) return index;
|
||||
index = parseUnicodeEscapeBackwards(s, index, Character::isWhitespace);
|
||||
if (index < 0) return -1;
|
||||
int nBackSlashes = 1;
|
||||
index--;
|
||||
if (index >= 0 && s.charAt(index) == '\\') {
|
||||
nBackSlashes++;
|
||||
nBackSlashes += countBackSlashes(s, index - 1);
|
||||
}
|
||||
return nBackSlashes % 2 == 0 ? -1 : index + 1;
|
||||
}
|
||||
|
||||
private static int countBackSlashes(@NotNull String s, int index) {
|
||||
int nBackSlashes = 0;
|
||||
while (index >= 0) {
|
||||
int start = s.charAt(index) == '\\' ? index : parseUnicodeEscapeBackwards(s, index, c -> c == '\\');
|
||||
if (start == -1) break;
|
||||
nBackSlashes++;
|
||||
index = start - 1;
|
||||
}
|
||||
return nBackSlashes;
|
||||
}
|
||||
|
||||
private static int parseUnicodeEscapeBackwards(@NotNull String s, int index, @NotNull Predicate<Character> charPredicate) {
|
||||
// \u1234 needs at least 6 positions
|
||||
if (index - 5 < 0) return -1;
|
||||
try {
|
||||
int code = Integer.parseInt(s.substring(index - 3, index + 1), 16);
|
||||
if (!charPredicate.test((char) code)) return -1;
|
||||
}
|
||||
catch (NumberFormatException e) {
|
||||
return -1;
|
||||
}
|
||||
if (s.charAt(index - 4) != 'u') return -1;
|
||||
index -= 5;
|
||||
// 'u' can appear multiple times
|
||||
while (index >= 0 && s.charAt(index) == 'u') index--;
|
||||
if (index < 0 || s.charAt(index) != '\\') return -1;
|
||||
return index;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -225,7 +225,16 @@ public final class JavaLexer extends LexerBase {
|
||||
if (pos >= myBufferEndOffset) return myBufferEndOffset;
|
||||
c = charAt(pos);
|
||||
if (c == '\n' || c == '\r') continue;
|
||||
pos++;
|
||||
if (c == 'u') {
|
||||
while (pos < myBufferEndOffset && charAt(pos) == 'u') pos++;
|
||||
if (pos + 3 >= myBufferEndOffset) return myBufferEndOffset;
|
||||
boolean isBackSlash = charAt(pos) == '0' && charAt(pos + 1) == '0' && charAt(pos + 2) == '5' && charAt(pos + 3) == 'c';
|
||||
// on encoded backslash we also need to skip escaped symbol (e.g. \\u005c" is translated to \")
|
||||
pos += (isBackSlash ? 5 : 4);
|
||||
}
|
||||
else {
|
||||
pos++;
|
||||
}
|
||||
if (pos >= myBufferEndOffset) return myBufferEndOffset;
|
||||
c = charAt(pos);
|
||||
}
|
||||
|
||||
+11
-1
@@ -28,7 +28,7 @@ public class a {
|
||||
};
|
||||
|
||||
String s1 = <error descr="Illegal escape character in string literal">"\xd"</error>;
|
||||
String s11= <error descr="Illegal escape character in string literal">"\udX"</error>;
|
||||
String s11= <error descr="Illegal line end in string literal">"\udX";</error><EOLError descr="';' expected"></EOLError>
|
||||
String s12= <error descr="Illegal escape character in string literal">"c:\TEMP\test.jar"</error>;
|
||||
String s3 = "";
|
||||
String s4 = "\u0000";
|
||||
@@ -41,6 +41,16 @@ public class a {
|
||||
String perverts = "\uuuuuuuuuuuu1234";
|
||||
char perv2 = '\uu3264';
|
||||
|
||||
String backSlash1 = <error descr="Illegal line end in string literal">"\u005c";</error><EOLError descr="';' expected"></EOLError>
|
||||
String backSlash2 = "\u005c\";
|
||||
String backSlash3 = "\\u005c";
|
||||
String backSlash4 = "\u005c\u005c";
|
||||
String backSlash5 = "\u005c134";
|
||||
String backSlash6 = <error descr="Illegal line end in string literal">"\134\u005c";</error><EOLError descr="';' expected"></EOLError>
|
||||
String backSlash7 = "\u005c\134";
|
||||
String backSlash8 = "\u005c\u0022";
|
||||
char backSlash9 = '\u005c\u0027';
|
||||
|
||||
void foo(String a) {
|
||||
foo(<error descr="Illegal line end in string literal">"aaa</error>
|
||||
);
|
||||
|
||||
@@ -15,4 +15,7 @@ class C {
|
||||
String valid2 = """
|
||||
\
|
||||
""";
|
||||
|
||||
String backSlash1 = """
|
||||
\u005c\""";
|
||||
}
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
class C {
|
||||
String unclosed = """
|
||||
\u005c""";
|
||||
}<EOLError descr="Unclosed text block"></EOLError><EOLError descr="'}' expected"></EOLError><EOLError descr="';' expected"></EOLError>
|
||||
+1
@@ -11,6 +11,7 @@ class JavaTextBlocksHighlightingTest : LightJavaCodeInsightFixtureTestCase() {
|
||||
|
||||
fun testTextBlocks() = doTest()
|
||||
fun testUnclosedTextBlock() = doTest()
|
||||
fun testUnclosedTextBlock2() = doTest()
|
||||
|
||||
fun testTextBlockOpeningSpaces() {
|
||||
myFixture.configureByText("${getTestName(false)}.java", "class C {\n String spaces = \"\"\" \t \u000C \n \"\"\";\n}")
|
||||
|
||||
@@ -15,6 +15,7 @@ import com.intellij.openapi.editor.highlighter.HighlighterIterator;
|
||||
import com.intellij.pom.java.LanguageLevel;
|
||||
import com.intellij.psi.JavaDocTokenType;
|
||||
import com.intellij.psi.JavaTokenType;
|
||||
import com.intellij.psi.StringEscapesTokenTypes;
|
||||
import com.intellij.testFramework.LightJavaCodeInsightTestCase;
|
||||
import com.intellij.testFramework.propertyBased.CheckHighlighterConsistency;
|
||||
|
||||
@@ -164,6 +165,30 @@ public class JavaHighlighterTest extends LightJavaCodeInsightTestCase {
|
||||
CheckHighlighterConsistency.performCheck(editor);
|
||||
}
|
||||
|
||||
public void testUnicodeEscapeSequence() {
|
||||
String prefix = "class A {\n" +
|
||||
" String s = \"\"\"\n";
|
||||
initDocument(prefix +
|
||||
"\\uuuuu005c\\\"\"\";\n" +
|
||||
"}");
|
||||
HighlighterIterator iterator = myHighlighter.createIterator(prefix.length());
|
||||
assertEquals(StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN, iterator.getTokenType());
|
||||
iterator.advance();
|
||||
assertEquals(JavaTokenType.TEXT_BLOCK_LITERAL, iterator.getTokenType());
|
||||
}
|
||||
|
||||
public void testUnicodeBackslashEscapesUnicodeSequence() {
|
||||
String prefix = "class A {\n" +
|
||||
" String s = \"\"\"\n";
|
||||
initDocument(prefix +
|
||||
"\\u005c\\u0040\"\"\";\n" +
|
||||
"}");
|
||||
HighlighterIterator iterator = myHighlighter.createIterator(prefix.length());
|
||||
assertEquals(StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN, iterator.getTokenType());
|
||||
iterator.advance();
|
||||
assertEquals(JavaTokenType.TEXT_BLOCK_LITERAL, iterator.getTokenType());
|
||||
}
|
||||
|
||||
private Editor initDocument(String text) {
|
||||
EditorFactory editorFactory = EditorFactory.getInstance();
|
||||
myDocument = editorFactory.createDocument(text);
|
||||
|
||||
@@ -1,17 +1,19 @@
|
||||
// Copyright 2000-2019 JetBrains s.r.o. Use of this source code is governed by the Apache 2.0 license that can be found in the LICENSE file.
|
||||
package com.intellij.psi.util;
|
||||
|
||||
import org.junit.Test;
|
||||
import com.intellij.psi.JavaPsiFacade;
|
||||
import com.intellij.psi.PsiElementFactory;
|
||||
import com.intellij.psi.PsiLiteralExpression;
|
||||
import com.intellij.testFramework.LightPlatformCodeInsightTestCase;
|
||||
import org.jetbrains.annotations.NotNull;
|
||||
|
||||
import static com.intellij.psi.util.PsiLiteralUtil.escapeBackSlashesInTextBlock;
|
||||
import static org.junit.Assert.assertEquals;
|
||||
|
||||
/**
|
||||
* @author Bas Leijdekkers
|
||||
*/
|
||||
public class PsiLiteralUtilTest {
|
||||
public class PsiLiteralUtilTest extends LightPlatformCodeInsightTestCase {
|
||||
|
||||
@Test
|
||||
public void testEscapeTextBlockCharacters() {
|
||||
assertEquals("foo \\s\n", PsiLiteralUtil.escapeTextBlockCharacters("foo \\n"));
|
||||
// escapes after 'bar' should be escaped since it's the last line in a text block
|
||||
@@ -31,11 +33,36 @@ public class PsiLiteralUtilTest {
|
||||
assertEquals("\\t\n", PsiLiteralUtil.escapeTextBlockCharacters("\\t\\n"));
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testEscapeBackSlashesInTextBlock() {
|
||||
assertEquals("", escapeBackSlashesInTextBlock(""));
|
||||
assertEquals("\\\\", escapeBackSlashesInTextBlock("\\"));
|
||||
// backslash before quote should be preserved
|
||||
assertEquals("\\\\\"", escapeBackSlashesInTextBlock("\\\""));
|
||||
}
|
||||
|
||||
public void testRemoveIncidentalWhitespacesInTextBlock() {
|
||||
assertEquals("", textBlockValue(" "));
|
||||
assertEquals("", textBlockValue("\\u0020"));
|
||||
assertEquals("", textBlockValue("\\uuuu0020"));
|
||||
assertEquals("", textBlockValue("\\u0020 "));
|
||||
assertEquals(" ", textBlockValue("\\040"));
|
||||
assertEquals("\\", textBlockValue("\\u005c\\\\u0020"));
|
||||
assertEquals("\\u0020", textBlockValue("\\\\u0020"));
|
||||
assertEquals("\\", textBlockValue("\\\\\\u0020"));
|
||||
assertEquals("\\\\u0020", textBlockValue("\\u005c\\u005c\\\\u0020"));
|
||||
assertEquals("", textBlockValue(" "));
|
||||
assertEquals("", textBlockValue(" "));
|
||||
assertEquals("", textBlockValue("\\u0009"));
|
||||
assertEquals("", textBlockValue("\\u0009 \\u0020 "));
|
||||
assertEquals("\\", textBlockValue("\\\\\\u0009"));
|
||||
}
|
||||
|
||||
private String textBlockValue(@NotNull String content) {
|
||||
PsiElementFactory factory = JavaPsiFacade.getElementFactory(getProject());
|
||||
String blockText = "\"\"\"\n" +
|
||||
content +
|
||||
"\"\"\"";
|
||||
PsiLiteralExpression textBlock = (PsiLiteralExpression)factory.createExpressionFromText(blockText, null);
|
||||
return (String)textBlock.getValue();
|
||||
}
|
||||
}
|
||||
@@ -29,6 +29,8 @@ import com.intellij.util.text.CharArrayUtil;
|
||||
import org.jetbrains.annotations.NotNull;
|
||||
import org.jetbrains.annotations.Nullable;
|
||||
|
||||
import java.util.Arrays;
|
||||
|
||||
public abstract class CodeInsightUtilCore extends FileModificationService {
|
||||
public static <T extends PsiElement> T findElementInRange(@NotNull PsiFile file,
|
||||
int startOffset,
|
||||
@@ -96,141 +98,218 @@ public abstract class CodeInsightUtilCore extends FileModificationService {
|
||||
return parseStringCharacters(chars, outChars, sourceOffsets, true, true, '"', '\'');
|
||||
}
|
||||
|
||||
public static boolean parseStringCharacters(@NotNull String chars, @NotNull StringBuilder outChars, int @Nullable [] sourceOffsets, boolean slashMustBeEscaped, boolean exitOnEscapingWrongSymbol, char @NotNull ... endChars) {
|
||||
assert sourceOffsets == null || sourceOffsets.length == chars.length()+1;
|
||||
if (chars.indexOf('\\') < 0) {
|
||||
outChars.append(chars);
|
||||
if (sourceOffsets != null) {
|
||||
for (int i = 0; i < sourceOffsets.length; i++) {
|
||||
sourceOffsets[i] = i;
|
||||
public static boolean parseStringCharacters(@NotNull String chars, @NotNull StringBuilder outChars,
|
||||
int @Nullable [] sourceOffsets, boolean slashMustBeEscaped,
|
||||
boolean exitOnEscapingWrongSymbol, char @NotNull ... endChars) {
|
||||
StringParser stringParser = new StringParser(sourceOffsets, slashMustBeEscaped, exitOnEscapingWrongSymbol, endChars);
|
||||
return stringParser.parse(chars, outChars);
|
||||
}
|
||||
|
||||
private static class StringParser {
|
||||
|
||||
private final int @Nullable [] mySourceOffsets;
|
||||
private final boolean mySlashMustBeEscaped;
|
||||
private final boolean myExitOnEscapingWrongSymbol;
|
||||
private final char @NotNull [] myEndChars;
|
||||
|
||||
private StringParser(int @Nullable [] sourceOffsets, boolean slashMustBeEscaped,
|
||||
boolean exitOnEscapingWrongSymbol, char @NotNull ... endChars) {
|
||||
mySourceOffsets = sourceOffsets;
|
||||
mySlashMustBeEscaped = slashMustBeEscaped;
|
||||
myExitOnEscapingWrongSymbol = exitOnEscapingWrongSymbol;
|
||||
myEndChars = endChars;
|
||||
}
|
||||
|
||||
private boolean parse(@NotNull String chars, @NotNull StringBuilder outChars) {
|
||||
assert mySourceOffsets == null || mySourceOffsets.length == chars.length() + 1;
|
||||
if (chars.indexOf('\\') < 0) {
|
||||
outChars.append(chars);
|
||||
if (mySourceOffsets != null) Arrays.setAll(mySourceOffsets, i -> i);
|
||||
return true;
|
||||
}
|
||||
int index = 0;
|
||||
final int outOffset = outChars.length();
|
||||
while (index < chars.length()) {
|
||||
char c = chars.charAt(index++);
|
||||
if (mySourceOffsets != null) {
|
||||
mySourceOffsets[outChars.length() - outOffset] = index - 1;
|
||||
mySourceOffsets[outChars.length() + 1 - outOffset] = index;
|
||||
}
|
||||
if (c != '\\') {
|
||||
outChars.append(c);
|
||||
continue;
|
||||
}
|
||||
index = parseEscapedSymbol(chars, outChars, index, outOffset, false);
|
||||
if (index == -1) return false;
|
||||
if (mySourceOffsets != null) {
|
||||
mySourceOffsets[outChars.length() - outOffset] = index;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
int index = 0;
|
||||
final int outOffset = outChars.length();
|
||||
while (index < chars.length()) {
|
||||
|
||||
private int parseEscapedSymbol(@NotNull String chars, @NotNull StringBuilder outChars,
|
||||
int index, int outOffset, boolean isAfterEscapedBackslash) {
|
||||
if (index == chars.length()) return -1;
|
||||
char c = chars.charAt(index++);
|
||||
if (sourceOffsets != null) {
|
||||
sourceOffsets[outChars.length()-outOffset] = index - 1;
|
||||
sourceOffsets[outChars.length() + 1 -outOffset] = index;
|
||||
if (parseEscapedChar(c, outChars)) {
|
||||
return index;
|
||||
}
|
||||
if (c != '\\') {
|
||||
outChars.append(c);
|
||||
continue;
|
||||
}
|
||||
if (index == chars.length()) return false;
|
||||
c = chars.charAt(index++);
|
||||
switch (c) {
|
||||
case'b':
|
||||
case '\\':
|
||||
boolean isUnicodeSequenceStart = isAfterEscapedBackslash && index < chars.length() && chars.charAt(index) == 'u';
|
||||
if (isUnicodeSequenceStart) {
|
||||
index = parseUnicodeEscape(chars, outChars, index, outOffset, true);
|
||||
}
|
||||
else {
|
||||
outChars.append('\\');
|
||||
}
|
||||
break;
|
||||
|
||||
case '0':
|
||||
case '1':
|
||||
case '2':
|
||||
case '3':
|
||||
case '4':
|
||||
case '5':
|
||||
case '6':
|
||||
case '7':
|
||||
index = parseOctalEscape(chars, outChars, c, index);
|
||||
break;
|
||||
|
||||
case 'u':
|
||||
if (isAfterEscapedBackslash) {
|
||||
if (!handleUnexpectedChar(outChars, index - 1, outOffset, c)) return -1;
|
||||
}
|
||||
else {
|
||||
index = parseUnicodeEscape(chars, outChars, index - 1, outOffset, false);
|
||||
}
|
||||
break;
|
||||
|
||||
default:
|
||||
if (!handleUnexpectedChar(outChars, index - 1, outOffset, c)) return -1;
|
||||
}
|
||||
return index;
|
||||
}
|
||||
|
||||
private int parseUnicodeEscape(@NotNull String s, @NotNull StringBuilder outChars, int index,
|
||||
int outOffset, boolean isAfterEscapedBackslash) {
|
||||
int len = s.length();
|
||||
int start = index - 1;
|
||||
// uuuuu1234 is valid too
|
||||
while (index < len && s.charAt(index) == 'u') {
|
||||
index++;
|
||||
}
|
||||
if (index + 4 > len) return -1;
|
||||
try {
|
||||
int code = Integer.parseInt(s.substring(index, index + 4), 16);
|
||||
//line separators are invalid here
|
||||
if (code == 0x000a || code == 0x000d) return -1;
|
||||
char c = s.charAt(index);
|
||||
if (c == '+' || c == '-') return -1;
|
||||
char escapedChar = (char)code;
|
||||
if (escapedChar == '\\') {
|
||||
if (isAfterEscapedBackslash) {
|
||||
// \u005c\u005c
|
||||
outChars.append('\\');
|
||||
return index + 4;
|
||||
}
|
||||
else {
|
||||
// u005cxyz
|
||||
return parseEscapedSymbol(s, outChars, index + 4, outOffset, true);
|
||||
}
|
||||
}
|
||||
if (isAfterEscapedBackslash) {
|
||||
// e.g. \u005c\u006e is converted to newline
|
||||
if (parseEscapedChar(escapedChar, outChars)) return index + 4;
|
||||
if (handleUnexpectedChar(outChars, start, outOffset, escapedChar)) return index + 4;
|
||||
return -1;
|
||||
}
|
||||
// just single unicode escape sequence
|
||||
outChars.append(escapedChar);
|
||||
return index + 4;
|
||||
}
|
||||
catch (NumberFormatException ignored) {
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
|
||||
private boolean handleUnexpectedChar(@NotNull StringBuilder outChars, int start, int outOffset, char c) {
|
||||
if (CharArrayUtil.indexOf(myEndChars, c, 0, myEndChars.length) != -1) {
|
||||
outChars.append(c);
|
||||
}
|
||||
else if (!myExitOnEscapingWrongSymbol) {
|
||||
if (!mySlashMustBeEscaped) {
|
||||
outChars.append('\\');
|
||||
if (mySourceOffsets != null) {
|
||||
mySourceOffsets[outChars.length() - outOffset] = start;
|
||||
}
|
||||
}
|
||||
outChars.append(c);
|
||||
}
|
||||
else {
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
private static boolean parseEscapedChar(char c, StringBuilder outChars) {
|
||||
switch (c) {
|
||||
case 'b':
|
||||
outChars.append('\b');
|
||||
break;
|
||||
return true;
|
||||
|
||||
case't':
|
||||
case 't':
|
||||
outChars.append('\t');
|
||||
break;
|
||||
return true;
|
||||
|
||||
case'n':
|
||||
case 'n':
|
||||
outChars.append('\n');
|
||||
break;
|
||||
return true;
|
||||
|
||||
case'f':
|
||||
case 'f':
|
||||
outChars.append('\f');
|
||||
break;
|
||||
return true;
|
||||
|
||||
case'r':
|
||||
case 'r':
|
||||
outChars.append('\r');
|
||||
break;
|
||||
return true;
|
||||
|
||||
case's':
|
||||
case 's':
|
||||
outChars.append(' ');
|
||||
break;
|
||||
return true;
|
||||
|
||||
case '\n':
|
||||
break;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
case'\\':
|
||||
outChars.append('\\');
|
||||
break;
|
||||
|
||||
case'0':
|
||||
case'1':
|
||||
case'2':
|
||||
case'3':
|
||||
case'4':
|
||||
case'5':
|
||||
case'6':
|
||||
case'7':
|
||||
char startC = c;
|
||||
int v = (int)c - '0';
|
||||
if (index < chars.length()) {
|
||||
c = chars.charAt(index++);
|
||||
private static int parseOctalEscape(@NotNull String s, @NotNull StringBuilder outChars, char c, int index) {
|
||||
char startC = c;
|
||||
int v = (int)c - '0';
|
||||
if (index < s.length()) {
|
||||
c = s.charAt(index++);
|
||||
if ('0' <= c && c <= '7') {
|
||||
v <<= 3;
|
||||
v += c - '0';
|
||||
if (startC <= '3' && index < s.length()) {
|
||||
c = s.charAt(index++);
|
||||
if ('0' <= c && c <= '7') {
|
||||
v <<= 3;
|
||||
v += c - '0';
|
||||
if (startC <= '3' && index < chars.length()) {
|
||||
c = chars.charAt(index++);
|
||||
if ('0' <= c && c <= '7') {
|
||||
v <<= 3;
|
||||
v += c - '0';
|
||||
}
|
||||
else {
|
||||
index--;
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
index--;
|
||||
}
|
||||
}
|
||||
outChars.append((char)v);
|
||||
break;
|
||||
|
||||
case'u':
|
||||
// uuuuu1234 is valid too
|
||||
while (index != chars.length() && chars.charAt(index) == 'u') {
|
||||
index++;
|
||||
}
|
||||
if (index + 4 <= chars.length()) {
|
||||
try {
|
||||
int code = Integer.parseInt(chars.substring(index, index + 4), 16);
|
||||
//line separators are invalid here
|
||||
if (code == 0x000a || code == 0x000d) return false;
|
||||
c = chars.charAt(index);
|
||||
if (c == '+' || c == '-') return false;
|
||||
outChars.append((char)code);
|
||||
index += 4;
|
||||
}
|
||||
catch (Exception e) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else {
|
||||
return false;
|
||||
}
|
||||
break;
|
||||
|
||||
default:
|
||||
if (CharArrayUtil.indexOf(endChars, c, 0, endChars.length) != -1) {
|
||||
outChars.append(c);
|
||||
}
|
||||
else if (!exitOnEscapingWrongSymbol) {
|
||||
if (!slashMustBeEscaped) {
|
||||
outChars.append('\\');
|
||||
if (sourceOffsets != null) {
|
||||
sourceOffsets[outChars.length() - outOffset] = index - 1;
|
||||
}
|
||||
}
|
||||
outChars.append(c);
|
||||
}
|
||||
else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (sourceOffsets != null) {
|
||||
sourceOffsets[outChars.length() - outOffset] = index;
|
||||
}
|
||||
else {
|
||||
index--;
|
||||
}
|
||||
}
|
||||
outChars.append((char)v);
|
||||
return index;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
+1
-1
@@ -12,7 +12,7 @@ class UnnecessaryUnicodeEscape {
|
||||
String str1 = "<warning descr="Unicode escape sequence '\u0061' can be replaced with 'a'">\u0061</warning>";
|
||||
String str2 = "\\u0061"; // Backslash followed by the characters "u0061"
|
||||
String str3 = "\\<warning descr="Unicode escape sequence '\u0061' can be replaced with 'a'">\u0061</warning>"; // Backslash followed by escape sequence
|
||||
String str4 = <error descr="Illegal escape character in string literal">"\u004"</error>; // Too short to be a Unicode escape sequence
|
||||
String str4 = <error descr="Illegal line end in string literal">"\u004"; // Too short to be a Unicode escape sequence</error><EOLError descr="';' expected"></EOLError>
|
||||
String str5 = <error descr="Illegal escape character in string literal">"\u004g"</error>; // Invalid hex character
|
||||
|
||||
// <warning descr="Unicode escape sequence '\u0009' can be replaced with a tab character">\u0009</warning>
|
||||
|
||||
Reference in New Issue
Block a user