mirror of
https://gitflic.ru/project/openide/openide.git
synced 2026-09-27 10:03:11 +07:00
recognize \N escape in Unicode string literals (PY-1313)
This commit is contained in:
@@ -5,6 +5,9 @@ import com.intellij.openapi.diagnostic.Logger;
|
||||
import com.intellij.psi.StringEscapesTokenTypes;
|
||||
import com.intellij.psi.tree.IElementType;
|
||||
|
||||
import static com.intellij.psi.StringEscapesTokenTypes.INVALID_UNICODE_ESCAPE_TOKEN;
|
||||
import static com.intellij.psi.StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN;
|
||||
|
||||
/**
|
||||
* Specialized lexer for string literals. To be used as a layer in a LayeredLexer.
|
||||
* Mostly handles escapes, differently in byte / unicode / raw strings.
|
||||
@@ -117,24 +120,33 @@ public class PyStringLiteralLexer extends LexerBase {
|
||||
char nextChar = myBuffer.charAt(myStart + 1);
|
||||
mySeenEscapedSpacesOnly &= nextChar == ' ';
|
||||
if ((nextChar == '\n' || nextChar == ' ' && (mySeenEscapedSpacesOnly || isTrailingSpace(myStart+2)))) {
|
||||
return StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN; // escaped EOL
|
||||
return VALID_STRING_ESCAPE_TOKEN; // escaped EOL
|
||||
}
|
||||
if (nextChar == 'u' || nextChar == 'U') {
|
||||
if (myUnicodeMark == MARK_UNICODE || (myUnicodeIsDefault && (myUnicodeMark == MARK_NONE))) { // unicode allowed
|
||||
if (isUnicodeMode()) {
|
||||
final int width = nextChar == 'u'? 4 : 8; // is it uNNNN or Unnnnnnnn
|
||||
for(int i = myStart + 2; i < myStart + width + 2; i++) {
|
||||
if (i >= myEnd || !isHexDigit(myBuffer.charAt(i))) return StringEscapesTokenTypes.INVALID_UNICODE_ESCAPE_TOKEN;
|
||||
if (i >= myEnd || !isHexDigit(myBuffer.charAt(i))) return INVALID_UNICODE_ESCAPE_TOKEN;
|
||||
}
|
||||
return StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN;
|
||||
return VALID_STRING_ESCAPE_TOKEN;
|
||||
}
|
||||
else return myOriginalLiteralToken; // b"\u1234" is just b"\\u1234", nothing gets escaped
|
||||
}
|
||||
|
||||
if (nextChar == 'x') { // \xNN is allowed both in bytes and unicode.
|
||||
for(int i = myStart + 2; i < myStart + 4; i++) {
|
||||
if (i >= myEnd || !isHexDigit(myBuffer.charAt(i))) return StringEscapesTokenTypes.INVALID_UNICODE_ESCAPE_TOKEN;
|
||||
if (i >= myEnd || !isHexDigit(myBuffer.charAt(i))) return INVALID_UNICODE_ESCAPE_TOKEN;
|
||||
}
|
||||
return StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN;
|
||||
return VALID_STRING_ESCAPE_TOKEN;
|
||||
}
|
||||
|
||||
if (nextChar == 'N' && isUnicodeMode()) {
|
||||
int i = myStart+2;
|
||||
if (i >= myEnd || myBuffer.charAt(i) != '{') return INVALID_UNICODE_ESCAPE_TOKEN;
|
||||
i++;
|
||||
while(i < myEnd && myBuffer.charAt(i) != '}') i++;
|
||||
if (i >= myEnd) return INVALID_UNICODE_ESCAPE_TOKEN;
|
||||
return VALID_STRING_ESCAPE_TOKEN;
|
||||
}
|
||||
|
||||
switch (nextChar) {
|
||||
@@ -155,13 +167,17 @@ public class PyStringLiteralLexer extends LexerBase {
|
||||
case '4':
|
||||
case '5':
|
||||
case '6':
|
||||
case '7': return StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN;
|
||||
case '7': return VALID_STRING_ESCAPE_TOKEN;
|
||||
}
|
||||
|
||||
// other unrecognized escapes are just part of string, not an error
|
||||
return myOriginalLiteralToken;
|
||||
}
|
||||
|
||||
private boolean isUnicodeMode() {
|
||||
return myUnicodeMark == MARK_UNICODE || (myUnicodeIsDefault && (myUnicodeMark == MARK_NONE));
|
||||
}
|
||||
|
||||
// all subsequent chars are escaped spaces
|
||||
private boolean isTrailingSpace(final int start) {
|
||||
for (int i=start; i<myBufferEnd; i+=2) {
|
||||
@@ -232,6 +248,18 @@ public class PyStringLiteralLexer extends LexerBase {
|
||||
}
|
||||
return i;
|
||||
}
|
||||
|
||||
if (myBuffer.charAt(i) == 'N' && isUnicodeMode()) {
|
||||
i++;
|
||||
while(i < myBufferEnd && myBuffer.charAt(i) != '}') {
|
||||
i++;
|
||||
}
|
||||
if (i < myBufferEnd) {
|
||||
i++;
|
||||
}
|
||||
return i;
|
||||
}
|
||||
|
||||
else {
|
||||
return i + 1;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,14 @@
|
||||
package com.jetbrains.python;
|
||||
|
||||
import com.jetbrains.python.fixtures.PyLexerTestCase;
|
||||
import com.jetbrains.python.lexer.PyStringLiteralLexer;
|
||||
|
||||
/**
|
||||
* @author yole
|
||||
*/
|
||||
public class PyStringLiteralLexerTest extends PyLexerTestCase {
|
||||
public void testBackslashN() { // PY-1313
|
||||
doLexerTest("u\"\\N{LATIN SMALL LETTER B}\"", new PyStringLiteralLexer(PyElementTypes.STRING_LITERAL_EXPRESSION, false),
|
||||
"Py:STRING_LITERAL_EXPRESSION", "VALID_STRING_ESCAPE_TOKEN", "Py:STRING_LITERAL_EXPRESSION");
|
||||
}
|
||||
}
|
||||
@@ -15,6 +15,7 @@ public class PythonAllTestsSuite {
|
||||
|
||||
public static final Class[] tests = {
|
||||
PythonLexerTest.class,
|
||||
PyStringLiteralLexerTest.class,
|
||||
PyEncodingTest.class,
|
||||
PythonParsingTest.class,
|
||||
PyStringLiteralTest.class,
|
||||
|
||||
@@ -1,13 +1,12 @@
|
||||
package com.jetbrains.python;
|
||||
|
||||
import com.intellij.lexer.Lexer;
|
||||
import com.jetbrains.python.fixtures.PyLexerTestCase;
|
||||
import com.jetbrains.python.lexer.PythonIndentingLexer;
|
||||
import junit.framework.TestCase;
|
||||
|
||||
/**
|
||||
* @author yole
|
||||
*/
|
||||
public class PythonLexerTest extends TestCase {
|
||||
public class PythonLexerTest extends PyLexerTestCase {
|
||||
public void testSimpleExpression() {
|
||||
doTest("a=1", "Py:IDENTIFIER", "Py:EQ", "Py:INTEGER_LITERAL");
|
||||
}
|
||||
@@ -156,28 +155,6 @@ public class PythonLexerTest extends TestCase {
|
||||
}
|
||||
|
||||
private static void doTest(String text, String... expectedTokens) {
|
||||
Lexer lexer = new PythonIndentingLexer();
|
||||
lexer.start(text);
|
||||
int idx = 0;
|
||||
int tokenPos = 0;
|
||||
while (lexer.getTokenType() != null) {
|
||||
if (idx >= expectedTokens.length) {
|
||||
StringBuilder remainingTokens = new StringBuilder(lexer.getTokenType().toString());
|
||||
lexer.advance();
|
||||
while (lexer.getTokenType() != null) {
|
||||
remainingTokens.append(" ").append(lexer.getTokenType().toString());
|
||||
lexer.advance();
|
||||
}
|
||||
fail("Too many tokens. Following tokens: " + remainingTokens.toString());
|
||||
}
|
||||
assertEquals("Token offset mismatch at position " + idx, tokenPos, lexer.getTokenStart());
|
||||
String tokenName = lexer.getTokenType().toString();
|
||||
assertEquals("Token mismatch at position " + idx, expectedTokens[idx], tokenName);
|
||||
idx++;
|
||||
tokenPos = lexer.getTokenEnd();
|
||||
lexer.advance();
|
||||
}
|
||||
|
||||
if (idx < expectedTokens.length) fail("Not enough tokens");
|
||||
doLexerTest(text, new PythonIndentingLexer(), expectedTokens);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,34 @@
|
||||
package com.jetbrains.python.fixtures;
|
||||
|
||||
import com.intellij.lexer.Lexer;
|
||||
import junit.framework.TestCase;
|
||||
|
||||
/**
|
||||
* @author yole
|
||||
*/
|
||||
public abstract class PyLexerTestCase extends TestCase {
|
||||
protected static void doLexerTest(String text, Lexer lexer, String... expectedTokens) {
|
||||
lexer.start(text);
|
||||
int idx = 0;
|
||||
int tokenPos = 0;
|
||||
while (lexer.getTokenType() != null) {
|
||||
if (idx >= expectedTokens.length) {
|
||||
StringBuilder remainingTokens = new StringBuilder(lexer.getTokenType().toString());
|
||||
lexer.advance();
|
||||
while (lexer.getTokenType() != null) {
|
||||
remainingTokens.append(" ").append(lexer.getTokenType().toString());
|
||||
lexer.advance();
|
||||
}
|
||||
fail("Too many tokens. Following tokens: " + remainingTokens.toString());
|
||||
}
|
||||
assertEquals("Token offset mismatch at position " + idx, tokenPos, lexer.getTokenStart());
|
||||
String tokenName = lexer.getTokenType().toString();
|
||||
assertEquals("Token mismatch at position " + idx, expectedTokens[idx], tokenName);
|
||||
idx++;
|
||||
tokenPos = lexer.getTokenEnd();
|
||||
lexer.advance();
|
||||
}
|
||||
|
||||
if (idx < expectedTokens.length) fail("Not enough tokens");
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user