PY-21697 Handle sequences like '\\''' in multiline strings in Python lexer

Previously, they would get matched as '\ followed by \', and then,
two remaining quotes. Therefore the literal was considered unterminated
by the lexer. It, in turn, broke multiple assertions in
PyStringLiteralLexer because it used to find different string's end than
PythonHighlightingLexer and rendered the whole editor unresponsive.
This commit is contained in:
Mikhail Golubev
2016-12-20 12:53:10 +03:00
parent 5ee2be25eb
commit ead37e8492
6 changed files with 88 additions and 54 deletions
@@ -195,14 +195,13 @@ class _PythonLexer implements FlexLexer {
"\1\116\1\117\1\0\1\45\1\120\1\121\1\0\1\120"+
"\1\3\1\47\2\122\1\0\2\123\1\0\1\3\1\124"+
"\1\125\10\3\1\126\1\3\1\127\3\3\1\0\2\121"+
"\6\0\2\104\1\0\2\3\1\130\1\3\1\131\1\132"+
"\4\3\1\133\1\3\1\134\1\0\2\120\2\0\2\122"+
"\1\0\3\104\1\135\1\136\1\137\1\3\1\140\1\141"+
"\1\3\1\142\3\120\3\122\1\123\1\0\1\143\2\0"+
"\1\3\1\144\4\0\2\104\1\145\2\120\2\122";
"\11\0\2\3\1\130\1\3\1\131\1\132\4\3\1\133"+
"\1\3\1\134\6\0\1\104\1\135\1\136\1\137\1\3"+
"\1\140\1\141\1\3\1\142\1\120\1\122\1\123\1\0"+
"\1\143\1\3\1\144\1\145";
private static int [] zzUnpackAction() {
int [] result = new int[297];
int [] result = new int[277];
int offset = 0;
offset = zzUnpackAction(ZZ_ACTION_PACKED_0, offset, result);
return result;
@@ -256,18 +255,15 @@ class _PythonLexer implements FlexLexer {
"\0\u2b12\0\u2b5c\0\u2ba6\0\u2bf0\0\u2c3a\0\u01bc\0\u01bc\0\u2c84"+
"\0\u2cce\0\u2d18\0\u2d62\0\u2dac\0\u2df6\0\u2e40\0\u2e8a\0\u01bc"+
"\0\u2ed4\0\u01bc\0\u2f1e\0\u2f68\0\u2fb2\0\u2ffc\0\u3046\0\u3090"+
"\0\u30da\0\u3124\0\u316e\0\u31b8\0\u3202\0\u324c\0\u3296\0\u32e0"+
"\0\u332a\0\u3374\0\u33be\0\u01bc\0\u3408\0\u01bc\0\u01bc\0\u3452"+
"\0\u349c\0\u34e6\0\u3530\0\u01bc\0\u357a\0\u01bc\0\u35c4\0\u360e"+
"\0\u3658\0\u36a2\0\u36ec\0\u3736\0\u3780\0\u37ca\0\u3814\0\u385e"+
"\0\u38a8\0\u01bc\0\u01bc\0\u01bc\0\u38f2\0\u01bc\0\u01bc\0\u393c"+
"\0\u01bc\0\u2956\0\u3986\0\u39d0\0\336\0\u3a1a\0\u3a64\0\336"+
"\0\u3814\0\336\0\u3aae\0\u3af8\0\u3b42\0\u01bc\0\u3b8c\0\u3bd6"+
"\0\u3c20\0\u3c6a\0\u324c\0\u332a\0\u01bc\0\u35c4\0\u36a2\0\u36ec"+
"\0\u37ca";
"\0\u30da\0\u3124\0\u316e\0\u31b8\0\u3202\0\u324c\0\u2b5c\0\u2ba6"+
"\0\u3296\0\u32e0\0\u332a\0\u01bc\0\u3374\0\u01bc\0\u01bc\0\u33be"+
"\0\u3408\0\u3452\0\u349c\0\u01bc\0\u34e6\0\u01bc\0\u3530\0\u3046"+
"\0\u3090\0\u357a\0\u35c4\0\u360e\0\u3658\0\u01bc\0\u01bc\0\u01bc"+
"\0\u36a2\0\u01bc\0\u01bc\0\u36ec\0\u01bc\0\u2956\0\336\0\336"+
"\0\u3658\0\336\0\u3736\0\u01bc\0\u01bc";
private static int [] zzUnpackRowMap() {
int [] result = new int[297];
int [] result = new int[277];
int offset = 0;
offset = zzUnpackRowMap(ZZ_ROWMAP_PACKED_0, offset, result);
return result;
@@ -525,40 +521,31 @@ class _PythonLexer implements FlexLexer {
"\1\0\3\7\10\0\1\375\25\7\30\0\12\7\2\0"+
"\2\7\1\0\1\7\1\0\3\7\10\0\4\7\1\376"+
"\21\7\27\0\25\311\1\377\1\u0100\175\311\140\314\1\u0101"+
"\1\u0102\62\314\25\317\1\u0103\1\u0104\175\317\140\320\1\u0105"+
"\1\u0106\62\320\25\250\1\u0107\1\357\110\250\1\u0108\1\357"+
"\63\250\26\253\1\360\1\u0109\110\253\1\360\1\u0107\62\253"+
"\1\0\12\7\2\0\2\7\1\0\1\7\1\0\3\7"+
"\10\0\6\7\1\u010a\17\7\30\0\12\7\2\0\2\7"+
"\1\0\1\7\1\0\3\7\10\0\6\7\1\u010b\17\7"+
"\1\u0102\62\314\25\317\1\u0103\1\353\175\317\140\320\1\354"+
"\1\u0104\62\320\25\250\1\u0105\1\357\63\250\26\253\1\360"+
"\1\u0105\62\253\1\0\12\7\2\0\2\7\1\0\1\7"+
"\1\0\3\7\10\0\6\7\1\u0106\17\7\30\0\12\7"+
"\2\0\2\7\1\0\1\7\1\0\3\7\10\0\6\7"+
"\1\u0107\17\7\30\0\12\7\2\0\2\7\1\0\1\7"+
"\1\0\3\7\10\0\1\7\1\u0108\24\7\30\0\12\7"+
"\2\0\2\7\1\0\1\7\1\0\3\7\10\0\1\7"+
"\1\u0109\24\7\30\0\12\7\2\0\2\7\1\0\1\7"+
"\1\0\3\7\10\0\1\u010a\25\7\30\0\12\7\2\0"+
"\2\7\1\0\1\7\1\0\3\7\10\0\6\7\1\u010b"+
"\17\7\30\0\12\7\2\0\2\7\1\0\1\7\1\0"+
"\3\7\10\0\12\7\1\u010c\13\7\30\0\12\7\2\0"+
"\2\7\1\0\1\7\1\0\3\7\10\0\12\7\1\u010d"+
"\13\7\27\0\25\311\1\u010e\1\u0100\63\311\26\314\1\u0101"+
"\1\u010e\62\314\25\317\1\u010f\1\353\63\317\26\320\1\354"+
"\1\u010f\62\320\26\0\1\u0110\1\0\2\u0111\1\0\2\u0112"+
"\56\0\12\7\2\0\2\7\1\0\1\7\1\0\3\7"+
"\10\0\15\7\1\u0113\10\7\30\0\12\7\2\0\2\7"+
"\1\0\1\7\1\0\3\7\10\0\21\7\1\u0114\4\7"+
"\30\0\12\7\2\0\2\7\1\0\1\7\1\0\3\7"+
"\10\0\1\7\1\u010c\24\7\30\0\12\7\2\0\2\7"+
"\1\0\1\7\1\0\3\7\10\0\1\7\1\u010d\24\7"+
"\30\0\12\7\2\0\2\7\1\0\1\7\1\0\3\7"+
"\10\0\1\u010e\25\7\30\0\12\7\2\0\2\7\1\0"+
"\1\7\1\0\3\7\10\0\6\7\1\u010f\17\7\30\0"+
"\12\7\2\0\2\7\1\0\1\7\1\0\3\7\10\0"+
"\12\7\1\u0110\13\7\30\0\12\7\2\0\2\7\1\0"+
"\1\7\1\0\3\7\10\0\12\7\1\u0111\13\7\27\0"+
"\25\311\1\u0112\1\u0100\110\311\1\u0113\1\u0100\63\311\26\314"+
"\1\u0101\1\u0114\110\314\1\u0101\1\u0112\62\314\25\317\1\u0115"+
"\1\u0104\110\317\1\u0116\1\u0104\63\317\26\320\1\u0105\1\u0117"+
"\110\320\1\u0105\1\u0115\62\320\26\0\1\u0118\1\0\2\u0119"+
"\1\0\2\u011a\55\0\25\250\1\u011b\1\357\63\250\26\253"+
"\1\360\1\u011c\62\253\1\0\12\7\2\0\2\7\1\0"+
"\1\7\1\0\3\7\10\0\15\7\1\u011d\10\7\30\0"+
"\12\7\2\0\2\7\1\0\1\7\1\0\3\7\10\0"+
"\21\7\1\u011e\4\7\27\0\25\311\1\u011f\1\u0100\63\311"+
"\26\314\1\u0101\1\u0120\62\314\25\317\1\u0121\1\u0104\63\317"+
"\26\320\1\u0105\1\u0122\62\320\25\250\1\u0123\1\357\63\250"+
"\26\253\1\360\1\u0124\62\253\1\0\12\7\2\0\2\7"+
"\1\0\1\7\1\0\3\7\10\0\4\7\1\u0125\21\7"+
"\27\0\25\311\1\u0126\1\u0100\63\311\26\314\1\u0101\1\u0127"+
"\62\314\25\317\1\u0128\1\u0104\63\317\26\320\1\u0105\1\u0129"+
"\62\320";
"\10\0\4\7\1\u0115\21\7\27\0";
private static int [] zzUnpackTrans() {
int [] result = new int[15540];
int [] result = new int[14208];
int offset = 0;
offset = zzUnpackTrans(ZZ_TRANS_PACKED_0, offset, result);
return result;
@@ -604,12 +591,11 @@ class _PythonLexer implements FlexLexer {
"\1\1\1\0\1\1\1\0\1\1\1\0\1\1\1\0"+
"\2\1\1\11\1\0\30\1\4\11\1\0\2\1\1\11"+
"\1\0\2\1\1\11\2\1\1\0\2\1\1\0\21\1"+
"\1\0\2\1\6\0\2\1\1\0\15\1\1\0\2\1"+
"\2\0\2\1\1\0\16\1\1\11\2\1\1\11\1\0"+
"\1\11\2\0\2\1\4\0\7\1";
"\1\0\2\1\11\0\15\1\6\0\12\1\2\11\1\0"+
"\1\11\3\1";
private static int [] zzUnpackAttribute() {
int [] result = new int[297];
int [] result = new int[277];
int offset = 0;
offset = zzUnpackAttribute(ZZ_ATTRIBUTE_PACKED_0, offset, result);
return result;
@@ -46,7 +46,7 @@ IMAGNUMBER=(({FLOATNUMBER})|({INTPART}))[Jj]
//RAW_STRING=[Rr]{QUOTED_STRING}
//QUOTED_STRING=({TRIPLE_APOS_LITERAL})|({QUOTED_LITERAL})|({DOUBLE_QUOTED_LITERAL})|({TRIPLE_QUOTED_LITERAL})
// If you change patterns for string literals, don't forget to update PythonStringUtil!
// If you change patterns for string literals, don't forget to update PyStringLiteralUtil!
// "c" prefix character is included for Cython
SINGLE_QUOTED_STRING=[UuBbCcRrFf]{0,3}({QUOTED_LITERAL} | {DOUBLE_QUOTED_LITERAL})
TRIPLE_QUOTED_STRING=[UuBbCcRrFf]{0,3}({TRIPLE_QUOTED_LITERAL}|{TRIPLE_APOS_LITERAL})
@@ -60,12 +60,12 @@ ESCAPE_SEQUENCE=\\[^\r\n]
ANY_ESCAPE_SEQUENCE = \\[^]
THREE_QUO = (\"\"\")
ONE_TWO_QUO = (\"[^\"]) | (\"\\[^]) | (\"\"[^\"]) | (\"\"\\[^])
ONE_TWO_QUO = (\"[^\\\"]) | (\"\\[^]) | (\"\"[^\\\"]) | (\"\"\\[^])
QUO_STRING_CHAR = [^\\\"] | {ANY_ESCAPE_SEQUENCE} | {ONE_TWO_QUO}
TRIPLE_QUOTED_LITERAL = {THREE_QUO} {QUO_STRING_CHAR}* {THREE_QUO}?
THREE_APOS = (\'\'\')
ONE_TWO_APOS = ('[^']) | ('\\[^]) | (''[^']) | (''\\[^])
ONE_TWO_APOS = ('[^\\']) | ('\\[^]) | (''[^\\']) | (''\\[^])
APOS_STRING_CHAR = [^\\'] | {ANY_ESCAPE_SEQUENCE} | {ONE_TWO_APOS}
TRIPLE_APOS_LITERAL = {THREE_APOS} {APOS_STRING_CHAR}* {THREE_APOS}?
@@ -0,0 +1,3 @@
s = '''
'\\''''
'''
@@ -0,0 +1,3 @@
s = '''
'\\<caret>''
'''
@@ -482,6 +482,12 @@ public class PyEditingTest extends PyTestCase {
"re.compile(ur'\\U00010000t')");
}
// PY-21697
public void testTripleQuotesInsideTripleQuotedStringLiteral() {
// TODO an extra quote is inserted due to PY-21993
doTypingTest("'");
}
private String doTestTyping(final String text, final int offset, final char character) {
final PsiFile file = WriteCommandAction.runWriteCommandAction(null, new Computable<PsiFile>() {
@Override
@@ -395,6 +395,42 @@ public class PythonLexerTest extends PyLexerTestCase {
doTest("s = f\"\"\"{x}\n\"\"\"", "Py:IDENTIFIER", "Py:SPACE", "Py:EQ", "Py:SPACE", "Py:TRIPLE_QUOTED_STRING", "Py:STATEMENT_BREAK");
}
// PY-21697
public void testTripleSingleQuotedStringWithEscapedSlashAfterOneQuote() {
doTest("s = '''\n" +
"'\\\\'''\n" +
"'''\n",
"Py:IDENTIFIER", "Py:SPACE", "Py:EQ", "Py:SPACE", "Py:TRIPLE_QUOTED_STRING",
"Py:STATEMENT_BREAK", "Py:LINE_BREAK", "Py:TRIPLE_QUOTED_STRING", "Py:STATEMENT_BREAK");
}
// PY-21697
public void testTripleSingleQuotedStringWithEscapedSlashAfterTwoQuotes() {
doTest("s = '''\n" +
"''\\\\'''\n" +
"'''\n",
"Py:IDENTIFIER", "Py:SPACE", "Py:EQ", "Py:SPACE", "Py:TRIPLE_QUOTED_STRING",
"Py:STATEMENT_BREAK", "Py:LINE_BREAK", "Py:TRIPLE_QUOTED_STRING", "Py:STATEMENT_BREAK");
}
// PY-21697
public void testTripleDoubleQuotedStringWithEscapedSlashAfterOneQuote() {
doTest("s = \"\"\"\n" +
"\"\\\\\"\"\"\n" +
"\"\"\"\n",
"Py:IDENTIFIER", "Py:SPACE", "Py:EQ", "Py:SPACE", "Py:TRIPLE_QUOTED_STRING",
"Py:STATEMENT_BREAK", "Py:LINE_BREAK", "Py:TRIPLE_QUOTED_STRING", "Py:STATEMENT_BREAK");
}
// PY-21697
public void testTripleDoubleQuotedStringWithEscapedSlashAfterTwoQuotes() {
doTest("s = \"\"\"\n" +
"\"\"\\\\\"\"\"\n" +
"\"\"\"\n",
"Py:IDENTIFIER", "Py:SPACE", "Py:EQ", "Py:SPACE", "Py:TRIPLE_QUOTED_STRING",
"Py:STATEMENT_BREAK", "Py:LINE_BREAK", "Py:TRIPLE_QUOTED_STRING", "Py:STATEMENT_BREAK");
}
private static void doTest(String text, String... expectedTokens) {
PyLexerTestCase.doLexerTest(text, new PythonIndentingLexer(), expectedTokens);
}