Fix lexing Java string literals containing a broken unicode escape (IDEA-293903)

GitOrigin-RevId: 3d2e29f7123b34b9f4837e008390636a01372802
This commit is contained in:
Bas Leijdekkers
2022-05-14 15:53:42 +00:00
committed by intellij-monorepo-bot
parent 4572d95998
commit f64ade1d8f
4 changed files with 97 additions and 61 deletions
@@ -8,23 +8,23 @@ import com.intellij.psi.TokenType;
import com.intellij.psi.impl.source.tree.JavaDocElementType;
import com.intellij.psi.tree.IElementType;
import com.intellij.util.containers.CollectionFactory;
import com.intellij.util.containers.ContainerUtil;
import com.intellij.util.text.CharArrayUtil;
import org.jetbrains.annotations.NotNull;
import org.jetbrains.annotations.Nullable;
import java.io.IOException;
import java.util.Arrays;
import java.util.HashSet;
import java.util.Set;
import static com.intellij.psi.PsiKeyword.*;
public final class JavaLexer extends LexerBase {
private static final Set<String> KEYWORDS = new HashSet<>(Arrays.asList(
private static final Set<String> KEYWORDS = ContainerUtil.set(
ABSTRACT, BOOLEAN, BREAK, BYTE, CASE, CATCH, CHAR, CLASS, CONST, CONTINUE, DEFAULT, DO, DOUBLE, ELSE, EXTENDS, FINAL, FINALLY,
FLOAT, FOR, GOTO, IF, IMPLEMENTS, IMPORT, INSTANCEOF, INT, INTERFACE, LONG, NATIVE, NEW, PACKAGE, PRIVATE, PROTECTED, PUBLIC,
RETURN, SHORT, STATIC, STRICTFP, SUPER, SWITCH, SYNCHRONIZED, THIS, THROW, THROWS, TRANSIENT, TRY, VOID, VOLATILE, WHILE,
TRUE, FALSE, NULL, NON_SEALED));
TRUE, FALSE, NULL, NON_SEALED);
private static final Set<CharSequence> JAVA9_KEYWORDS = CollectionFactory.createCharSequenceSet(Arrays.asList(OPEN, MODULE, REQUIRES, EXPORTS, OPENS, USES, PROVIDES, TRANSITIVE, TO, WITH));
@@ -51,6 +51,9 @@ public final class JavaLexer extends LexerBase {
private int myTokenEndOffset; // positioned after the last symbol of the current token
private IElementType myTokenType;
/** The length of the last valid unicode escape (6 or greater), or 1 when no unicode escape was found. */
private int mySymbolLength = 1;
public JavaLexer(@NotNull LanguageLevel level) {
myFlexLexer = new _JavaLexer(level);
}
@@ -63,6 +66,7 @@ public final class JavaLexer extends LexerBase {
myBufferEndOffset = endOffset;
myTokenType = null;
myTokenEndOffset = startOffset;
mySymbolLength = 1;
myFlexLexer.reset(myBuffer, startOffset, endOffset, 0);
}
@@ -156,7 +160,7 @@ public final class JavaLexer extends LexerBase {
break;
case '\'':
myTokenType = JavaTokenType.CHARACTER_LITERAL;
myTokenEndOffset = getClosingQuote(myBufferIndex + 1, c);
myTokenEndOffset = getClosingQuote(myBufferIndex + 1, '\'');
break;
case '"':
@@ -166,7 +170,7 @@ public final class JavaLexer extends LexerBase {
}
else {
myTokenType = JavaTokenType.STRING_LITERAL;
myTokenEndOffset = getClosingQuote(myBufferIndex + 1, c);
myTokenEndOffset = getClosingQuote(myBufferIndex + 1, '"');
}
break;
@@ -180,17 +184,12 @@ public final class JavaLexer extends LexerBase {
}
private int getWhitespaces(int offset) {
if (offset >= myBufferEndOffset) {
return myBufferEndOffset;
}
int pos = offset;
char c = charAt(pos);
while (c == ' ' || c == '\t' || c == '\n' || c == '\r' || c == '\f') {
while (pos < myBufferEndOffset) {
char c = charAt(pos);
if (c != ' ' && c != '\t' && c != '\n' && c != '\r' && c != '\f') break;
pos++;
if (pos == myBufferEndOffset) return pos;
c = charAt(pos);
}
return pos;
@@ -206,61 +205,33 @@ public final class JavaLexer extends LexerBase {
}
private int getClosingQuote(int offset, char quoteChar) {
if (offset >= myBufferEndOffset) {
return myBufferEndOffset;
}
int pos = offset;
char c = charAt(pos);
while (true) {
while (c != quoteChar && c != '\n' && c != '\r' && c != '\\') {
pos++;
if (pos >= myBufferEndOffset) return myBufferEndOffset;
c = charAt(pos);
}
while (pos < myBufferEndOffset) {
char c = charAt(pos);
if (c == '\\') {
pos++;
if (pos >= myBufferEndOffset) return myBufferEndOffset;
c = charAt(pos);
if (c == '\n' || c == '\r') continue;
if (c == 'u') {
do {
pos++;
}
while (pos < myBufferEndOffset && charAt(pos) == 'u');
if (pos + 3 >= myBufferEndOffset) return myBufferEndOffset;
boolean isBackSlash = charAt(pos) == '0' && charAt(pos + 1) == '0' && charAt(pos + 2) == '5' && charAt(pos + 3) == 'c';
// on encoded backslash we also need to skip escaped symbol (e.g. \\u005c" is translated to \")
pos += (isBackSlash ? 5 : 4);
}
else {
pos++;
}
if (pos >= myBufferEndOffset) return myBufferEndOffset;
c = charAt(pos);
char d = symbolAt(pos);
pos += mySymbolLength;
if (d != '\\') continue;
// on (encoded) backslash we also need to skip the next symbol (e.g. \\u005c" is translated to \")
}
else if (c == quoteChar) {
break;
return pos + 1;
}
else {
pos--;
break;
else if (c == '\n' || c == '\r') {
return pos;
}
pos++;
}
return pos + 1;
return pos;
}
private int getClosingComment(int offset) {
int pos = offset;
while (pos < myBufferEndOffset - 1) {
char c = charAt(pos);
if (c == '*' && (charAt(pos + 1)) == '/') {
break;
}
if (charAt(pos) == '*' && (charAt(pos + 1)) == '/') break;
pos++;
}
@@ -283,13 +254,12 @@ public final class JavaLexer extends LexerBase {
int pos = offset;
while ((pos = getClosingQuote(pos + 1, '"')) < myBufferEndOffset) {
char current = charAt(pos);
if (current == '\\') {
char c = charAt(pos);
if (c == '\\') {
pos++;
}
else if (current == '"' && pos + 1 < myBufferEndOffset && charAt(pos + 1) == '"') {
pos += 2;
break;
else if (c == '"' && pos + 1 < myBufferEndOffset && charAt(pos + 1) == '"') {
return pos + 2;
}
}
@@ -300,6 +270,28 @@ public final class JavaLexer extends LexerBase {
return myBufferArray != null ? myBufferArray[position] : myBuffer.charAt(position);
}
private char symbolAt(int offset) {
mySymbolLength = 1;
int pos = offset;
char first = charAt(pos);
if (first != '\\') return first;
if (++pos >= myBufferEndOffset || charAt(pos) != 'u') return first;
//noinspection StatementWithEmptyBody
while (++pos < myBufferEndOffset && charAt(pos) == 'u');
if (pos + 3 >= myBufferEndOffset) return first;
int result = 0;
for (int max = pos + 4; pos < max; pos++) {
result <<= 4;
char c = charAt(pos);
if ('0' <= c && c <= '9') result += c - '0';
else if ('a' <= c && c <= 'f') result += (c - 'a') + 10;
else if ('A' <= c && c <= 'F') result += (c - 'A') + 10;
else return first;
}
mySymbolLength = pos - offset;
return (char)result;
}
@NotNull
@Override
public CharSequence getBufferSequence() {
@@ -28,7 +28,7 @@ public class a {
};
String s1 = <error descr="Illegal escape character in string literal">"\xd"</error>;
String s11= <error descr="Illegal line end in string literal">"\udX";</error><EOLError descr="';' expected"></EOLError>
String s11= <error descr="Illegal escape character in string literal">"\udX"</error>;
String s12= <error descr="Illegal escape character in string literal">"c:\TEMP\test.jar"</error>;
String s3 = "";
String s4 = "\u0000";
@@ -1,4 +1,4 @@
// Copyright 2000-2019 JetBrains s.r.o. Use of this source code is governed by the Apache 2.0 license that can be found in the LICENSE file.
// Copyright 2000-2022 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
package com.intellij.java.lexer;
import com.intellij.lang.java.JavaParserDefinition;
@@ -158,6 +158,50 @@ public class JavaLexerTest extends LexerTestCase {
doTest("\"\"\"\n \"\\\"\"\" \"\"\" ", "TEXT_BLOCK_LITERAL ('\"\"\"\\n \"\\\"\"\" \"\"\"')\nWHITE_SPACE (' ')");
doTest("\"\"\"\n \"\"\\\"\"\" \"\"\" ", "TEXT_BLOCK_LITERAL ('\"\"\"\\n \"\"\\\"\"\" \"\"\"')\nWHITE_SPACE (' ')");
doTest("\"\"\" \n\"\"\" ", "TEXT_BLOCK_LITERAL ('\"\"\" \\n\"\"\"')\nWHITE_SPACE (' ')");
doTest("\"\"\"\n \\u005C\"\"\"\n \"\"\"", "TEXT_BLOCK_LITERAL ('\"\"\"\\n \\u005C\"\"\"\\n \"\"\"')"); // unicode escaped backslash '\'
}
public void testStringLiterals() {
doTest("\"", "STRING_LITERAL ('\"')");
doTest("\" ", "STRING_LITERAL ('\" ')");
doTest("\"\"", "STRING_LITERAL ('\"\"')");
doTest("\"\" ", "STRING_LITERAL ('\"\"')\nWHITE_SPACE (' ')");
doTest("\"\\\"\" ", "STRING_LITERAL ('\"\\\"\"')\nWHITE_SPACE (' ')");
doTest("\"\\", "STRING_LITERAL ('\"\\')");
doTest("\"\\u", "STRING_LITERAL ('\"\\u')");
doTest("\"\n\"", "STRING_LITERAL ('\"')\nWHITE_SPACE ('\\n')\nSTRING_LITERAL ('\"')");
doTest("\"\\n\" ", "STRING_LITERAL ('\"\\n\"')\nWHITE_SPACE (' ')");
doTest("\"\\u005c\"\" ", "STRING_LITERAL ('\"\\u005c\"\"')\nWHITE_SPACE (' ')");
doTest("\"\\u005\" ", "STRING_LITERAL ('\"\\u005\"')\nWHITE_SPACE (' ')"); // broken unicode escape
doTest("\"\\u00\" ", "STRING_LITERAL ('\"\\u00\"')\nWHITE_SPACE (' ')"); // broken unicode escape
doTest("\"\\u0\" ", "STRING_LITERAL ('\"\\u0\"')\nWHITE_SPACE (' ')"); // broken unicode escape
doTest("\"\\u\" ", "STRING_LITERAL ('\"\\u\"')\nWHITE_SPACE (' ')"); // broken unicode escape
}
public void testCharLiterals() {
doTest("'\\u005c\\u005c'", "CHARACTER_LITERAL (''\\u005c\\u005c'')"); // unicode escaped escaped slash '\\'
doTest("'\\u1234' ", "CHARACTER_LITERAL (''\\u1234'')\nWHITE_SPACE (' ')");
doTest("'x' ", "CHARACTER_LITERAL (''x'')\nWHITE_SPACE (' ')");
doTest("'", "CHARACTER_LITERAL (''')");
doTest("' ", "CHARACTER_LITERAL ('' ')");
doTest("''", "CHARACTER_LITERAL ('''')");
doTest("'\\u007F' 'F' ", "CHARACTER_LITERAL (''\\u007F'')\nWHITE_SPACE (' ')\nCHARACTER_LITERAL (''F'')\nWHITE_SPACE (' ')");
doTest("'\\u005C' ", "CHARACTER_LITERAL (''\\u005C' ')"); // closing quote is escaped with unicode escaped backslash
}
public void testComments() {
doTest("//", "END_OF_LINE_COMMENT ('//')");
doTest("//\n", "END_OF_LINE_COMMENT ('//')\nWHITE_SPACE ('\\n')");
doTest("//x\n", "END_OF_LINE_COMMENT ('//x')\nWHITE_SPACE ('\\n')");
doTest("/*/ ", "C_STYLE_COMMENT ('/*/ ')");
doTest("/**/ ", "C_STYLE_COMMENT ('/**/')\nWHITE_SPACE (' ')");
doTest("/*x*/ ", "C_STYLE_COMMENT ('/*x*/')\nWHITE_SPACE (' ')");
doTest("/***/ ", "DOC_COMMENT ('/***/')\nWHITE_SPACE (' ')");
doTest("/**x*/ ", "DOC_COMMENT ('/**x*/')\nWHITE_SPACE (' ')");
doTest("/*", "C_STYLE_COMMENT ('/*')");
doTest("/**", "DOC_COMMENT ('/**')");
doTest("/***", "DOC_COMMENT ('/***')");
doTest("#! ", "END_OF_LINE_COMMENT ('#! ')");
}
@Override
@@ -12,7 +12,7 @@ class UnnecessaryUnicodeEscape {
String str1 = "<warning descr="Unicode escape sequence '\u0061' can be replaced with 'a'">\u0061</warning>";
String str2 = "\\u0061"; // Backslash followed by the characters "u0061"
String str3 = "\\<warning descr="Unicode escape sequence '\u0061' can be replaced with 'a'">\u0061</warning>"; // Backslash followed by escape sequence
String str4 = <error descr="Illegal line end in string literal">"\u004"; // Too short to be a Unicode escape sequence</error><EOLError descr="';' expected"></EOLError>
String str4 = <error descr="Illegal escape character in string literal">"\u004"</error>; // Too short to be a Unicode escape sequence
String str5 = <error descr="Illegal escape character in string literal">"\u004g"</error>; // Invalid hex character
// <warning descr="Unicode escape sequence '\u0009' can be replaced with a tab character">\u0009</warning>