From 6bcf1ea6230fbe541f9c9880def79d0d8ed349cc Mon Sep 17 00:00:00 2001 From: Mikhail Golubev Date: Fri, 11 Nov 2016 16:13:57 +0300 Subject: [PATCH] PY-21399 Backslash in the middle of an escape sequence ends it in PyStringLiteralLexer Otherwise it might miss the following escaped quote and stop scanning a string literal earlier than the actual Python lexer. --- .../python/lexer/PyStringLiteralLexer.java | 8 ++++---- .../python/PyStringLiteralLexerTest.java | 18 ++++++++++++++++++ .../python/fixtures/PyLexerTestCase.java | 9 +++++---- 3 files changed, 27 insertions(+), 8 deletions(-) diff --git a/python/src/com/jetbrains/python/lexer/PyStringLiteralLexer.java b/python/src/com/jetbrains/python/lexer/PyStringLiteralLexer.java index 733146696903..5fb00af58f65 100644 --- a/python/src/com/jetbrains/python/lexer/PyStringLiteralLexer.java +++ b/python/src/com/jetbrains/python/lexer/PyStringLiteralLexer.java @@ -226,7 +226,7 @@ public class PyStringLiteralLexer extends LexerBase { if (myBuffer.charAt(i) == 'x') { i++; for (; i < start + 4; i++) { - if (i == myBufferEnd || myBuffer.charAt(i) == '\n' || myBuffer.charAt(i) == myQuoteChar) { + if (i == myBufferEnd || myBuffer.charAt(i) == '\n' || myBuffer.charAt(i) == myQuoteChar || myBuffer.charAt(i) == '\\') { return i; } } @@ -238,7 +238,7 @@ public class PyStringLiteralLexer extends LexerBase { final int width = myBuffer.charAt(i) == 'u'? 4 : 8; // is it uNNNN or Unnnnnnnn i++; for (; i < start + width + 2; i++) { - if (i == myBufferEnd || myBuffer.charAt(i) == '\n' || myBuffer.charAt(i) == myQuoteChar) { + if (i == myBufferEnd || myBuffer.charAt(i) == '\n' || myBuffer.charAt(i) == myQuoteChar || myBuffer.charAt(i) == '\\') { return i; } } @@ -247,10 +247,10 @@ public class PyStringLiteralLexer extends LexerBase { if (myBuffer.charAt(i) == 'N' && isUnicodeMode()) { i++; - while(i < myBufferEnd && myBuffer.charAt(i) != '}') { + while(i < myBufferEnd && myBuffer.charAt(i) != '}' && myBuffer.charAt(i) != '\\') { i++; } - if (i < myBufferEnd) { + if (i < myBufferEnd && myBuffer.charAt(i) == '}') { i++; } return i; diff --git a/python/testSrc/com/jetbrains/python/PyStringLiteralLexerTest.java b/python/testSrc/com/jetbrains/python/PyStringLiteralLexerTest.java index b51a5d05d95b..cc516fb8a9f0 100644 --- a/python/testSrc/com/jetbrains/python/PyStringLiteralLexerTest.java +++ b/python/testSrc/com/jetbrains/python/PyStringLiteralLexerTest.java @@ -38,4 +38,22 @@ public class PyStringLiteralLexerTest extends PyLexerTestCase { PyLexerTestCase.doLexerTest("fff'foo'", new PyStringLiteralLexer(PyTokenTypes.SINGLE_QUOTED_UNICODE), "Py:SINGLE_QUOTED_UNICODE"); PyLexerTestCase.doLexerTest("rrr'''foo'''", new PyStringLiteralLexer(PyTokenTypes.TRIPLE_QUOTED_UNICODE), "Py:TRIPLE_QUOTED_UNICODE"); } + + // PY-21399 + public void testBackslashBreaksAnyEscapeSequence() { + PyLexerTestCase.doLexerTest("'\\u\\')", new PyStringLiteralLexer(PyTokenTypes.SINGLE_QUOTED_UNICODE), true, "'", "\\u", "\\'", ")"); + PyLexerTestCase.doLexerTest("'\\u\\')", new PyStringLiteralLexer(PyTokenTypes.SINGLE_QUOTED_UNICODE), + "Py:SINGLE_QUOTED_UNICODE", "INVALID_UNICODE_ESCAPE_TOKEN", "VALID_STRING_ESCAPE_TOKEN", + "Py:SINGLE_QUOTED_UNICODE"); + + PyLexerTestCase.doLexerTest("'\\x\\')", new PyStringLiteralLexer(PyTokenTypes.SINGLE_QUOTED_UNICODE), true, "'", "\\x", "\\'", ")"); + PyLexerTestCase.doLexerTest("'\\x\\')", new PyStringLiteralLexer(PyTokenTypes.SINGLE_QUOTED_UNICODE), + "Py:SINGLE_QUOTED_UNICODE", "INVALID_UNICODE_ESCAPE_TOKEN", "VALID_STRING_ESCAPE_TOKEN", + "Py:SINGLE_QUOTED_UNICODE"); + + PyLexerTestCase.doLexerTest("'\\N{F\\')", new PyStringLiteralLexer(PyTokenTypes.SINGLE_QUOTED_UNICODE), true, "'", "\\N{F", "\\'", ")"); + PyLexerTestCase.doLexerTest("'\\N{F\\')", new PyStringLiteralLexer(PyTokenTypes.SINGLE_QUOTED_UNICODE), + "Py:SINGLE_QUOTED_UNICODE", "INVALID_UNICODE_ESCAPE_TOKEN", "VALID_STRING_ESCAPE_TOKEN", + "Py:SINGLE_QUOTED_UNICODE"); + } } diff --git a/python/testSrc/com/jetbrains/python/fixtures/PyLexerTestCase.java b/python/testSrc/com/jetbrains/python/fixtures/PyLexerTestCase.java index bdef84359fd3..ede46d362040 100644 --- a/python/testSrc/com/jetbrains/python/fixtures/PyLexerTestCase.java +++ b/python/testSrc/com/jetbrains/python/fixtures/PyLexerTestCase.java @@ -46,11 +46,12 @@ public abstract class PyLexerTestCase extends PlatformLiteFixture { int tokenPos = 0; while (lexer.getTokenType() != null) { if (idx >= expectedTokens.length) { - StringBuilder remainingTokens = new StringBuilder("\"" + lexer.getTokenType().toString() + "\""); - lexer.advance(); + final StringBuilder remainingTokens = new StringBuilder(); while (lexer.getTokenType() != null) { - remainingTokens.append(","); - remainingTokens.append(" \"").append(checkTokenText ? lexer.getTokenText() : lexer.getTokenType().toString()).append("\""); + if (remainingTokens.length() != 0) { + remainingTokens.append(", "); + } + remainingTokens.append("\"").append(checkTokenText ? lexer.getTokenText() : lexer.getTokenType().toString()).append("\""); lexer.advance(); } fail("Too many tokens. Following tokens: " + remainingTokens.toString());