PY-21840 Consider surrogate pairs in Python's StringLiteralTextEscaper

Namely, detect a single unsplittable piece of a string returned by
getDecodedFragments() by checking whether the decoded text contains
only one code point, not code unit (as returned by String.length()),
since some escape sequences, in particular, long unicode escape
sequences might be decoded into a surrogate pair.
This commit is contained in:
Mikhail Golubev
2016-12-20 12:53:10 +03:00
parent 6c6ca898fb
commit 5ee2be25eb
3 changed files with 29 additions and 4 deletions
@@ -351,7 +351,7 @@ public class PyStringLiteralExpressionImpl extends PyElementImpl implements PySt
if (intersection != null && !intersection.isEmpty()) {
final String value = fragment.getSecond();
final String intersectedValue;
if (value.length() == 1 || value.length() == intersection.getLength()) {
if (value.codePointCount(0, value.length()) == 1 || value.length() == intersection.getLength()) {
intersectedValue = value;
}
else {
@@ -367,7 +367,7 @@ public class PyStringLiteralExpressionImpl extends PyElementImpl implements PySt
@Override
public int getOffsetInHost(final int offsetInDecoded, @NotNull final TextRange rangeInsideHost) {
int offset = 0;
int offset = 0; // running offset in the decoded fragment
int endOffset = -1;
for (Pair<TextRange, String> fragment : myHost.getDecodedFragments()) {
final TextRange encodedTextRange = fragment.getFirst();
@@ -379,13 +379,15 @@ public class PyStringLiteralExpressionImpl extends PyElementImpl implements PySt
if (valueLength == 0) {
return -1;
}
else if (valueLength == 1) {
// A long unicode escape of form \U01234567 can be decoded into a surrogate pair
else if (value.codePointCount(0, valueLength) == 1) {
if (offset == offsetInDecoded) {
return intersection.getStartOffset();
}
offset++;
offset += valueLength;
}
else {
// Literal fragment without escapes: it's safe to use intersection length instead of value length
if (offset + intersectionLength >= offsetInDecoded) {
final int delta = offsetInDecoded - offset;
return intersection.getStartOffset() + delta;
@@ -472,6 +472,16 @@ public class PyEditingTest extends PyTestCase {
")");
}
// PY-21840
public void testEditInjectedRegexpFragmentWithLongUnicodeEscape() {
myFixture.configureByText(PythonFileType.INSTANCE,
"import re\n" +
"re.compile(ur'\\U00010000<caret>')");
doTyping("t");
myFixture.checkResult("import re\n" +
"re.compile(ur'\\U00010000t')");
}
private String doTestTyping(final String text, final int offset, final char character) {
final PsiFile file = WriteCommandAction.runWriteCommandAction(null, new Computable<PsiFile>() {
@Override
@@ -115,6 +115,19 @@ public class PyStringLiteralTest extends PyTestCase {
assertEquals(-1, escaper.getOffsetInHost(9, range));
}
public void testEscaperOffsetInLongUnicodeEscape() {
final PyStringLiteralExpression expr = createLiteralFromText("u'XXX a\\U0001F600\\U0001F600b YYY'");
final LiteralTextEscaper<? extends PsiLanguageInjectionHost> escaper = expr.createLiteralTextEscaper();
final TextRange range = TextRange.create(6, 28);
assertEquals(6, escaper.getOffsetInHost(0, range));
assertEquals(7, escaper.getOffsetInHost(1, range));
// Each \\U0001F600 is represented as a surrogate pair, hence 2 characters-wide step in decoded text
assertEquals(17, escaper.getOffsetInHost(3, range));
assertEquals(27, escaper.getOffsetInHost(5, range));
assertEquals(28, escaper.getOffsetInHost(6, range));
assertEquals(-1, escaper.getOffsetInHost(7, range));
}
public void testStringValue() {
assertEquals("foo", createLiteralFromText("\"\"\"foo\"\"\"").getStringValue());
assertEquals("foo", createLiteralFromText("u\"foo\"").getStringValue());