Java: lex nested string templates correctly (IDEA-321503)

GitOrigin-RevId: f88d2034128a7a1dcae5f1adca5a9105c6109eb6
This commit is contained in:
Bas Leijdekkers
2023-08-01 11:58:22 +00:00
committed by intellij-monorepo-bot
parent bd4417426a
commit d50045f34f
3 changed files with 82 additions and 11 deletions
@@ -55,7 +55,8 @@ public final class JavaLexer extends LexerBase {
private int myTokenEndOffset; // positioned after the last symbol of the current token
private IElementType myTokenType;
private short myState = 0;
private short myBraceCount = 0;
private short[] myBraceCounts = new short[1];
private int myTemplateDepth = -1;
/** The length of the last valid unicode escape (6 or greater), or 1 when no unicode escape was found. */
private int mySymbolLength = 1;
@@ -76,14 +77,18 @@ public final class JavaLexer extends LexerBase {
mySymbolLength = 1;
if (initialState != 0) {
myState = (short)initialState;
myBraceCount = (short)(initialState >> 16);
short braceCount = (short)(initialState >> 16);
if (braceCount > 0) {
myBraceCounts[0] = braceCount;
myTemplateDepth = 0;
}
}
myFlexLexer.reset(myBuffer, startOffset, endOffset, 0);
}
@Override
public int getState() {
return (myBraceCount == 0) ? 0 : myBraceCount << 16 | myState;
return myTemplateDepth < 0 ? 0 : myBraceCounts[myTemplateDepth] << 16 | myState;
}
@Override
@@ -135,16 +140,16 @@ public final class JavaLexer extends LexerBase {
break;
case '{':
if (myBraceCount > 0) {
myBraceCount++;
if (myTemplateDepth >= 0) {
myBraceCounts[myTemplateDepth]++;
}
myTokenType = JavaTokenType.LBRACE;
myTokenEndOffset = myBufferIndex + mySymbolLength;
break;
case '}':
if (myBraceCount > 0) {
myBraceCount--;
if (myBraceCount == 0) {
if (myTemplateDepth >= 0) {
if (--myBraceCounts[myTemplateDepth] == 0) {
myTemplateDepth--;
if (myState == STATE_TEXT_BLOCK_TEMPLATE) {
if (locateTextBlockEnd(myBufferIndex + mySymbolLength)) {
myTokenType = JavaTokenType.TEXT_BLOCK_TEMPLATE_MID;
@@ -287,7 +292,11 @@ public final class JavaLexer extends LexerBase {
if (pos < myBufferEndOffset) {
if (locateCharAt(pos) == '{' && myStringTemplates && quoteChar == '"') {
pos += mySymbolLength;
myBraceCount = 1;
myTemplateDepth++;
if (myTemplateDepth == myBraceCounts.length) {
myBraceCounts = Arrays.copyOf(myBraceCounts, myTemplateDepth * 2);
}
myBraceCounts[myTemplateDepth] = 1;
myTokenEndOffset = pos;
return true;
}
@@ -36,4 +36,8 @@ class X {
System.out.println(<error descr="Raw processor type is not allowed: X.MyProcessor">myProcessor</error>."");
return <error descr="Raw processor type is not allowed: java.lang.StringTemplate.Processor">processor</error>."\{}\{}\{}\{}\{}\{}";
}
void nested() {
System.out.println(STR."\{STR."\{STR."\{STR."\{STR."\{STR."\{STR.""}"}"}"}"}"}");
}
}
@@ -333,7 +333,64 @@ public class JavaLexerTest extends LexerTestCase {
WHITE_SPACE (' ')
TEXT_BLOCK_TEMPLATE_END ('}""\"')
WHITE_SPACE (' ')""");
//doTest("\"\\{fruit[0]}, \\{STR.\"\\{fruit[1]}, \\{fruit[2]}\"}\"","");
doTest("""
"\\{fruit[0]}, \\{STR."\\{fruit[1]}, \\{fruit[2]}"}"
""",
"""
STRING_TEMPLATE_BEGIN ('"\\{')
IDENTIFIER ('fruit')
LBRACKET ('[')
INTEGER_LITERAL ('0')
RBRACKET (']')
STRING_TEMPLATE_MID ('}, \\{')
IDENTIFIER ('STR')
DOT ('.')
STRING_TEMPLATE_BEGIN ('"\\{')
IDENTIFIER ('fruit')
LBRACKET ('[')
INTEGER_LITERAL ('1')
RBRACKET (']')
STRING_TEMPLATE_MID ('}, \\{')
IDENTIFIER ('fruit')
LBRACKET ('[')
INTEGER_LITERAL ('2')
RBRACKET (']')
STRING_TEMPLATE_END ('}"')
STRING_TEMPLATE_END ('}"')
WHITE_SPACE ('\\n')""");
doTest("""
STR."\\{STR."\\{STR."\\{STR."\\{STR."\\{STR."\\{STR.""}"}"}"}"}"}"
""",
"""
IDENTIFIER ('STR')
DOT ('.')
STRING_TEMPLATE_BEGIN ('"\\{')
IDENTIFIER ('STR')
DOT ('.')
STRING_TEMPLATE_BEGIN ('"\\{')
IDENTIFIER ('STR')
DOT ('.')
STRING_TEMPLATE_BEGIN ('"\\{')
IDENTIFIER ('STR')
DOT ('.')
STRING_TEMPLATE_BEGIN ('"\\{')
IDENTIFIER ('STR')
DOT ('.')
STRING_TEMPLATE_BEGIN ('"\\{')
IDENTIFIER ('STR')
DOT ('.')
STRING_TEMPLATE_BEGIN ('"\\{')
IDENTIFIER ('STR')
DOT ('.')
STRING_LITERAL ('""')
STRING_TEMPLATE_END ('}"')
STRING_TEMPLATE_END ('}"')
STRING_TEMPLATE_END ('}"')
STRING_TEMPLATE_END ('}"')
STRING_TEMPLATE_END ('}"')
STRING_TEMPLATE_END ('}"')
WHITE_SPACE ('\\n')
""");
}
public void testStringLiterals() {
@@ -404,7 +461,8 @@ public class JavaLexerTest extends LexerTestCase {
doTest("/", "DIV ('/')");
doTest("1/2", "INTEGER_LITERAL ('1')\nDIV ('/')\nINTEGER_LITERAL ('2')");
doTest("//\\\\u000A test", "END_OF_LINE_COMMENT ('//\\\\u000A test')"); // escaped backslash, not a unicode escape
doTest("//\\\\\\u000A test", "END_OF_LINE_COMMENT ('//\\\\')\nWHITE_SPACE ('\\u000A ')\nIDENTIFIER ('test')"); // escaped backslash, followed by a unicode escape
doTest("//\\\\\\u000A test",
"END_OF_LINE_COMMENT ('//\\\\')\nWHITE_SPACE ('\\u000A ')\nIDENTIFIER ('test')"); // escaped backslash, followed by a unicode escape
}
public void testWhitespace() {