recognize \N escape in Unicode string literals (PY-1313)

This commit is contained in:
Dmitry Jemerov
2010-07-19 17:19:49 +04:00
parent d9dd5f88d0
commit 206bf65ba1
5 changed files with 87 additions and 33 deletions
@@ -5,6 +5,9 @@ import com.intellij.openapi.diagnostic.Logger;
import com.intellij.psi.StringEscapesTokenTypes;
import com.intellij.psi.tree.IElementType;
import static com.intellij.psi.StringEscapesTokenTypes.INVALID_UNICODE_ESCAPE_TOKEN;
import static com.intellij.psi.StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN;
/**
* Specialized lexer for string literals. To be used as a layer in a LayeredLexer.
* Mostly handles escapes, differently in byte / unicode / raw strings.
@@ -117,24 +120,33 @@ public class PyStringLiteralLexer extends LexerBase {
char nextChar = myBuffer.charAt(myStart + 1);
mySeenEscapedSpacesOnly &= nextChar == ' ';
if ((nextChar == '\n' || nextChar == ' ' && (mySeenEscapedSpacesOnly || isTrailingSpace(myStart+2)))) {
return StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN; // escaped EOL
return VALID_STRING_ESCAPE_TOKEN; // escaped EOL
}
if (nextChar == 'u' || nextChar == 'U') {
if (myUnicodeMark == MARK_UNICODE || (myUnicodeIsDefault && (myUnicodeMark == MARK_NONE))) { // unicode allowed
if (isUnicodeMode()) {
final int width = nextChar == 'u'? 4 : 8; // is it uNNNN or Unnnnnnnn
for(int i = myStart + 2; i < myStart + width + 2; i++) {
if (i >= myEnd || !isHexDigit(myBuffer.charAt(i))) return StringEscapesTokenTypes.INVALID_UNICODE_ESCAPE_TOKEN;
if (i >= myEnd || !isHexDigit(myBuffer.charAt(i))) return INVALID_UNICODE_ESCAPE_TOKEN;
}
return StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN;
return VALID_STRING_ESCAPE_TOKEN;
}
else return myOriginalLiteralToken; // b"\u1234" is just b"\\u1234", nothing gets escaped
}
if (nextChar == 'x') { // \xNN is allowed both in bytes and unicode.
for(int i = myStart + 2; i < myStart + 4; i++) {
if (i >= myEnd || !isHexDigit(myBuffer.charAt(i))) return StringEscapesTokenTypes.INVALID_UNICODE_ESCAPE_TOKEN;
if (i >= myEnd || !isHexDigit(myBuffer.charAt(i))) return INVALID_UNICODE_ESCAPE_TOKEN;
}
return StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN;
return VALID_STRING_ESCAPE_TOKEN;
}
if (nextChar == 'N' && isUnicodeMode()) {
int i = myStart+2;
if (i >= myEnd || myBuffer.charAt(i) != '{') return INVALID_UNICODE_ESCAPE_TOKEN;
i++;
while(i < myEnd && myBuffer.charAt(i) != '}') i++;
if (i >= myEnd) return INVALID_UNICODE_ESCAPE_TOKEN;
return VALID_STRING_ESCAPE_TOKEN;
}
switch (nextChar) {
@@ -155,13 +167,17 @@ public class PyStringLiteralLexer extends LexerBase {
case '4':
case '5':
case '6':
case '7': return StringEscapesTokenTypes.VALID_STRING_ESCAPE_TOKEN;
case '7': return VALID_STRING_ESCAPE_TOKEN;
}
// other unrecognized escapes are just part of string, not an error
return myOriginalLiteralToken;
}
private boolean isUnicodeMode() {
return myUnicodeMark == MARK_UNICODE || (myUnicodeIsDefault && (myUnicodeMark == MARK_NONE));
}
// all subsequent chars are escaped spaces
private boolean isTrailingSpace(final int start) {
for (int i=start; i<myBufferEnd; i+=2) {
@@ -232,6 +248,18 @@ public class PyStringLiteralLexer extends LexerBase {
}
return i;
}
if (myBuffer.charAt(i) == 'N' && isUnicodeMode()) {
i++;
while(i < myBufferEnd && myBuffer.charAt(i) != '}') {
i++;
}
if (i < myBufferEnd) {
i++;
}
return i;
}
else {
return i + 1;
}
@@ -0,0 +1,14 @@
package com.jetbrains.python;
import com.jetbrains.python.fixtures.PyLexerTestCase;
import com.jetbrains.python.lexer.PyStringLiteralLexer;
/**
* @author yole
*/
public class PyStringLiteralLexerTest extends PyLexerTestCase {
public void testBackslashN() { // PY-1313
doLexerTest("u\"\\N{LATIN SMALL LETTER B}\"", new PyStringLiteralLexer(PyElementTypes.STRING_LITERAL_EXPRESSION, false),
"Py:STRING_LITERAL_EXPRESSION", "VALID_STRING_ESCAPE_TOKEN", "Py:STRING_LITERAL_EXPRESSION");
}
}
@@ -15,6 +15,7 @@ public class PythonAllTestsSuite {
public static final Class[] tests = {
PythonLexerTest.class,
PyStringLiteralLexerTest.class,
PyEncodingTest.class,
PythonParsingTest.class,
PyStringLiteralTest.class,
@@ -1,13 +1,12 @@
package com.jetbrains.python;
import com.intellij.lexer.Lexer;
import com.jetbrains.python.fixtures.PyLexerTestCase;
import com.jetbrains.python.lexer.PythonIndentingLexer;
import junit.framework.TestCase;
/**
* @author yole
*/
public class PythonLexerTest extends TestCase {
public class PythonLexerTest extends PyLexerTestCase {
public void testSimpleExpression() {
doTest("a=1", "Py:IDENTIFIER", "Py:EQ", "Py:INTEGER_LITERAL");
}
@@ -156,28 +155,6 @@ public class PythonLexerTest extends TestCase {
}
private static void doTest(String text, String... expectedTokens) {
Lexer lexer = new PythonIndentingLexer();
lexer.start(text);
int idx = 0;
int tokenPos = 0;
while (lexer.getTokenType() != null) {
if (idx >= expectedTokens.length) {
StringBuilder remainingTokens = new StringBuilder(lexer.getTokenType().toString());
lexer.advance();
while (lexer.getTokenType() != null) {
remainingTokens.append(" ").append(lexer.getTokenType().toString());
lexer.advance();
}
fail("Too many tokens. Following tokens: " + remainingTokens.toString());
}
assertEquals("Token offset mismatch at position " + idx, tokenPos, lexer.getTokenStart());
String tokenName = lexer.getTokenType().toString();
assertEquals("Token mismatch at position " + idx, expectedTokens[idx], tokenName);
idx++;
tokenPos = lexer.getTokenEnd();
lexer.advance();
}
if (idx < expectedTokens.length) fail("Not enough tokens");
doLexerTest(text, new PythonIndentingLexer(), expectedTokens);
}
}
@@ -0,0 +1,34 @@
package com.jetbrains.python.fixtures;
import com.intellij.lexer.Lexer;
import junit.framework.TestCase;
/**
* @author yole
*/
public abstract class PyLexerTestCase extends TestCase {
protected static void doLexerTest(String text, Lexer lexer, String... expectedTokens) {
lexer.start(text);
int idx = 0;
int tokenPos = 0;
while (lexer.getTokenType() != null) {
if (idx >= expectedTokens.length) {
StringBuilder remainingTokens = new StringBuilder(lexer.getTokenType().toString());
lexer.advance();
while (lexer.getTokenType() != null) {
remainingTokens.append(" ").append(lexer.getTokenType().toString());
lexer.advance();
}
fail("Too many tokens. Following tokens: " + remainingTokens.toString());
}
assertEquals("Token offset mismatch at position " + idx, tokenPos, lexer.getTokenStart());
String tokenName = lexer.getTokenType().toString();
assertEquals("Token mismatch at position " + idx, expectedTokens[idx], tokenName);
idx++;
tokenPos = lexer.getTokenEnd();
lexer.advance();
}
if (idx < expectedTokens.length) fail("Not enough tokens");
}
}