diff --git a/python/src/com/jetbrains/python/lexer/PyStringLiteralLexer.java b/python/src/com/jetbrains/python/lexer/PyStringLiteralLexer.java index aed42d0d9eee..6a49da32b61b 100644 --- a/python/src/com/jetbrains/python/lexer/PyStringLiteralLexer.java +++ b/python/src/com/jetbrains/python/lexer/PyStringLiteralLexer.java @@ -5,6 +5,9 @@ import com.intellij.openapi.diagnostic.Logger; import com.intellij.psi.StringEscapesTokenTypes; import com.intellij.psi.tree.IElementType; +import static com.intellij.psi.StringEscapesTokenTypes.INVALID_UNICODE_ESCAPE_TOKEN; +import static com.intellij.psi.StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN; + /** * Specialized lexer for string literals. To be used as a layer in a LayeredLexer. * Mostly handles escapes, differently in byte / unicode / raw strings. @@ -117,24 +120,33 @@ public class PyStringLiteralLexer extends LexerBase { char nextChar = myBuffer.charAt(myStart + 1); mySeenEscapedSpacesOnly &= nextChar == ' '; if ((nextChar == '\n' || nextChar == ' ' && (mySeenEscapedSpacesOnly || isTrailingSpace(myStart+2)))) { - return StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN; // escaped EOL + return VALID_STRING_ESCAPE_TOKEN; // escaped EOL } if (nextChar == 'u' || nextChar == 'U') { - if (myUnicodeMark == MARK_UNICODE || (myUnicodeIsDefault && (myUnicodeMark == MARK_NONE))) { // unicode allowed + if (isUnicodeMode()) { final int width = nextChar == 'u'? 4 : 8; // is it uNNNN or Unnnnnnnn for(int i = myStart + 2; i < myStart + width + 2; i++) { - if (i >= myEnd || !isHexDigit(myBuffer.charAt(i))) return StringEscapesTokenTypes.INVALID_UNICODE_ESCAPE_TOKEN; + if (i >= myEnd || !isHexDigit(myBuffer.charAt(i))) return INVALID_UNICODE_ESCAPE_TOKEN; } - return StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN; + return VALID_STRING_ESCAPE_TOKEN; } else return myOriginalLiteralToken; // b"\u1234" is just b"\\u1234", nothing gets escaped } if (nextChar == 'x') { // \xNN is allowed both in bytes and unicode. for(int i = myStart + 2; i < myStart + 4; i++) { - if (i >= myEnd || !isHexDigit(myBuffer.charAt(i))) return StringEscapesTokenTypes.INVALID_UNICODE_ESCAPE_TOKEN; + if (i >= myEnd || !isHexDigit(myBuffer.charAt(i))) return INVALID_UNICODE_ESCAPE_TOKEN; } - return StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN; + return VALID_STRING_ESCAPE_TOKEN; + } + + if (nextChar == 'N' && isUnicodeMode()) { + int i = myStart+2; + if (i >= myEnd || myBuffer.charAt(i) != '{') return INVALID_UNICODE_ESCAPE_TOKEN; + i++; + while(i < myEnd && myBuffer.charAt(i) != '}') i++; + if (i >= myEnd) return INVALID_UNICODE_ESCAPE_TOKEN; + return VALID_STRING_ESCAPE_TOKEN; } switch (nextChar) { @@ -155,13 +167,17 @@ public class PyStringLiteralLexer extends LexerBase { case '4': case '5': case '6': - case '7': return StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN; + case '7': return VALID_STRING_ESCAPE_TOKEN; } // other unrecognized escapes are just part of string, not an error return myOriginalLiteralToken; } + private boolean isUnicodeMode() { + return myUnicodeMark == MARK_UNICODE || (myUnicodeIsDefault && (myUnicodeMark == MARK_NONE)); + } + // all subsequent chars are escaped spaces private boolean isTrailingSpace(final int start) { for (int i=start; i= expectedTokens.length) { - StringBuilder remainingTokens = new StringBuilder(lexer.getTokenType().toString()); - lexer.advance(); - while (lexer.getTokenType() != null) { - remainingTokens.append(" ").append(lexer.getTokenType().toString()); - lexer.advance(); - } - fail("Too many tokens. Following tokens: " + remainingTokens.toString()); - } - assertEquals("Token offset mismatch at position " + idx, tokenPos, lexer.getTokenStart()); - String tokenName = lexer.getTokenType().toString(); - assertEquals("Token mismatch at position " + idx, expectedTokens[idx], tokenName); - idx++; - tokenPos = lexer.getTokenEnd(); - lexer.advance(); - } - - if (idx < expectedTokens.length) fail("Not enough tokens"); + doLexerTest(text, new PythonIndentingLexer(), expectedTokens); } } diff --git a/python/testSrc/com/jetbrains/python/fixtures/PyLexerTestCase.java b/python/testSrc/com/jetbrains/python/fixtures/PyLexerTestCase.java new file mode 100644 index 000000000000..d41382b192e8 --- /dev/null +++ b/python/testSrc/com/jetbrains/python/fixtures/PyLexerTestCase.java @@ -0,0 +1,34 @@ +package com.jetbrains.python.fixtures; + +import com.intellij.lexer.Lexer; +import junit.framework.TestCase; + +/** + * @author yole + */ +public abstract class PyLexerTestCase extends TestCase { + protected static void doLexerTest(String text, Lexer lexer, String... expectedTokens) { + lexer.start(text); + int idx = 0; + int tokenPos = 0; + while (lexer.getTokenType() != null) { + if (idx >= expectedTokens.length) { + StringBuilder remainingTokens = new StringBuilder(lexer.getTokenType().toString()); + lexer.advance(); + while (lexer.getTokenType() != null) { + remainingTokens.append(" ").append(lexer.getTokenType().toString()); + lexer.advance(); + } + fail("Too many tokens. Following tokens: " + remainingTokens.toString()); + } + assertEquals("Token offset mismatch at position " + idx, tokenPos, lexer.getTokenStart()); + String tokenName = lexer.getTokenType().toString(); + assertEquals("Token mismatch at position " + idx, expectedTokens[idx], tokenName); + idx++; + tokenPos = lexer.getTokenEnd(); + lexer.advance(); + } + + if (idx < expectedTokens.length) fail("Not enough tokens"); + } +}