PY-21399 Backslash in the middle of an escape sequence ends it in PyStringLiteralLexer

Otherwise it might miss the following escaped quote and stop scanning
a string literal earlier than the actual Python lexer.
This commit is contained in:
Mikhail Golubev
2016-11-11 17:43:25 +03:00
parent 0bd39935dd
commit 6bcf1ea623
3 changed files with 27 additions and 8 deletions
@@ -38,4 +38,22 @@ public class PyStringLiteralLexerTest extends PyLexerTestCase {
PyLexerTestCase.doLexerTest("fff'foo'", new PyStringLiteralLexer(PyTokenTypes.SINGLE_QUOTED_UNICODE), "Py:SINGLE_QUOTED_UNICODE");
PyLexerTestCase.doLexerTest("rrr'''foo'''", new PyStringLiteralLexer(PyTokenTypes.TRIPLE_QUOTED_UNICODE), "Py:TRIPLE_QUOTED_UNICODE");
}
// PY-21399
public void testBackslashBreaksAnyEscapeSequence() {
PyLexerTestCase.doLexerTest("'\\u\\')", new PyStringLiteralLexer(PyTokenTypes.SINGLE_QUOTED_UNICODE), true, "'", "\\u", "\\'", ")");
PyLexerTestCase.doLexerTest("'\\u\\')", new PyStringLiteralLexer(PyTokenTypes.SINGLE_QUOTED_UNICODE),
"Py:SINGLE_QUOTED_UNICODE", "INVALID_UNICODE_ESCAPE_TOKEN", "VALID_STRING_ESCAPE_TOKEN",
"Py:SINGLE_QUOTED_UNICODE");
PyLexerTestCase.doLexerTest("'\\x\\')", new PyStringLiteralLexer(PyTokenTypes.SINGLE_QUOTED_UNICODE), true, "'", "\\x", "\\'", ")");
PyLexerTestCase.doLexerTest("'\\x\\')", new PyStringLiteralLexer(PyTokenTypes.SINGLE_QUOTED_UNICODE),
"Py:SINGLE_QUOTED_UNICODE", "INVALID_UNICODE_ESCAPE_TOKEN", "VALID_STRING_ESCAPE_TOKEN",
"Py:SINGLE_QUOTED_UNICODE");
PyLexerTestCase.doLexerTest("'\\N{F\\')", new PyStringLiteralLexer(PyTokenTypes.SINGLE_QUOTED_UNICODE), true, "'", "\\N{F", "\\'", ")");
PyLexerTestCase.doLexerTest("'\\N{F\\')", new PyStringLiteralLexer(PyTokenTypes.SINGLE_QUOTED_UNICODE),
"Py:SINGLE_QUOTED_UNICODE", "INVALID_UNICODE_ESCAPE_TOKEN", "VALID_STRING_ESCAPE_TOKEN",
"Py:SINGLE_QUOTED_UNICODE");
}
}
@@ -46,11 +46,12 @@ public abstract class PyLexerTestCase extends PlatformLiteFixture {
int tokenPos = 0;
while (lexer.getTokenType() != null) {
if (idx >= expectedTokens.length) {
StringBuilder remainingTokens = new StringBuilder("\"" + lexer.getTokenType().toString() + "\"");
lexer.advance();
final StringBuilder remainingTokens = new StringBuilder();
while (lexer.getTokenType() != null) {
remainingTokens.append(",");
remainingTokens.append(" \"").append(checkTokenText ? lexer.getTokenText() : lexer.getTokenType().toString()).append("\"");
if (remainingTokens.length() != 0) {
remainingTokens.append(", ");
}
remainingTokens.append("\"").append(checkTokenText ? lexer.getTokenText() : lexer.getTokenType().toString()).append("\"");
lexer.advance();
}
fail("Too many tokens. Following tokens: " + remainingTokens.toString());