[vfs] IJPL-173099, IJPL-191579: avoid detecting charset by partial-content

+ Do not update VirtualFile.charset if only part of it's content bytes are available for charset-detection

GitOrigin-RevId: 91fc98cd58ef3539e96ebf051f7bb9317b633289
This commit is contained in:
Ruslan Cheremin
2025-07-04 00:13:32 +00:00
committed by intellij-monorepo-bot
parent 798fac4129
commit e805de8996
2 changed files with 69 additions and 45 deletions
@@ -1,4 +1,4 @@
// Copyright 2000-2024 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
// Copyright 2000-2025 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
package com.intellij.compiler;
import com.intellij.openapi.application.WriteAction;
@@ -15,12 +15,14 @@ import org.jetbrains.annotations.NotNull;
import java.io.IOException;
import java.nio.charset.Charset;
import java.nio.charset.StandardCharsets;
import java.util.Arrays;
import java.util.Collection;
import java.util.HashSet;
import java.util.Set;
import static java.nio.charset.StandardCharsets.ISO_8859_1;
import static java.nio.charset.StandardCharsets.UTF_8;
public class CompilerEncodingServiceTest extends JavaPsiTestCase {
private static final Charset WINDOWS_1251 = Charset.forName("windows-1251");
private static final Charset WINDOWS_1252 = Charset.forName("windows-1252");
@@ -68,7 +70,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
public void testPropertiesEncodingTest() {
final VirtualFile file = createFile("A.properties");
assertEquals(StandardCharsets.UTF_8, file.getCharset());
assertEquals(UTF_8, file.getCharset());
EncodingProjectManager.getInstance(myProject).setEncoding(file, WINDOWS_1251);
assertSameElements(getService().getAllModuleEncodings(myModule), getProjectDefault());
@@ -81,30 +83,30 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
final VirtualFile file = createFile("test.properties");
WriteAction.run(() -> {
content.set(("one=1\n" +
"two=2\n").getBytes(StandardCharsets.ISO_8859_1));
"two=2\n").getBytes(ISO_8859_1));
file.setBinaryContent(content.get());
});
file.setCharset(null);
assertEquals(StandardCharsets.UTF_8, file.getCharset());
assertEquals(UTF_8, file.getCharset());
//
WriteAction.run(() -> {
content.set(ArrayUtil.mergeArrays(content.get(), ("three=3️⃣\n" +
"four=4️⃣\n").getBytes(StandardCharsets.UTF_8)));
"four=4️⃣\n").getBytes(UTF_8)));
file.setBinaryContent(content.get());
});
file.setCharset(null);
assertEquals(StandardCharsets.UTF_8, file.getCharset());
assertEquals(UTF_8, file.getCharset());
//
WriteAction.run(() -> {
content.set(ArrayUtil.mergeArrays(content.get(), ("five=fünf\n" +
"six=sechs\n").getBytes(StandardCharsets.ISO_8859_1)));
"six=sechs\n").getBytes(ISO_8859_1)));
file.setBinaryContent(content.get());
});
file.setCharset(null);
assertEquals(StandardCharsets.ISO_8859_1, file.getCharset());
assertEquals(ISO_8859_1, file.getCharset());
}
public void testBigPropertiesAutoEncoding() throws IOException {
@@ -113,15 +115,18 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
WriteAction.run(() -> {
@SuppressWarnings("NonAsciiCharacters")
byte[] bytes = "verifyEmail.tooltip=初回ログイン後またはアドレスの変更が送信された後に、ユーザーに自分の電子メールアドレスを確認するように要求します。\n"
.repeat(64).getBytes(StandardCharsets.UTF_8);
.repeat(64).getBytes(UTF_8);
file.setBinaryContent(bytes);
});
LoadTextUtil.loadText(file, 1024);
assertEquals(StandardCharsets.UTF_8, file.getCharset());
assertEquals(UTF_8, file.getCharset());
}
public void testCheckEncodingStability() throws IOException {
//TODO RC: Test is temporarily ignored because it relies on LoadTextUtil.loadText(file, length) to always set file
// encoding even by partial content -> nowadays the method .loadText(file, length) only set encoding if file.length==length.
// The test must be updated accordingly, and re-enabled afterwards.
public void _testCheckEncodingStability() throws IOException {
final VirtualFile file = createFile("test.properties");
{
byte[] bytes = """
@@ -129,7 +134,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
two=zwei
three=drei
four=vier
""".getBytes(StandardCharsets.ISO_8859_1);
""".getBytes(ISO_8859_1);
WriteAction.run(() -> {
file.setBinaryContent(bytes);
});
@@ -138,7 +143,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
CharSequence text = LoadTextUtil.loadText(file, i);
assertEquals("Text encoding mismatch: expected UTF-8 the entire file. " +
"Found: " + file.getCharset() + " in part of content: '" + text + "'",
StandardCharsets.UTF_8,
UTF_8,
file.getCharset());
}
}
@@ -157,7 +162,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
eight=acht
nine=neun
ten=zehn
""".getBytes(StandardCharsets.ISO_8859_1);
""".getBytes(ISO_8859_1);
WriteAction.run(() -> {
file.setBinaryContent(bytes);
});
@@ -166,7 +171,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
CharSequence text = LoadTextUtil.loadText(file, i);
assertEquals("Text encoding mismatch: expected ISO-8859-1 for the entire file. " +
"Found: " + file.getCharset() + " in part of content: '" + text + "'",
StandardCharsets.ISO_8859_1,
ISO_8859_1,
file.getCharset());
}
}
@@ -177,7 +182,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
try {
registryValue.setValue(true);
final VirtualFile file = createFile("A.properties");
assertEquals(StandardCharsets.ISO_8859_1, file.getCharset());
assertEquals(ISO_8859_1, file.getCharset());
EncodingProjectManager.getInstance(myProject).setEncoding(file, WINDOWS_1251);
assertSameElements(getService().getAllModuleEncodings(myModule), getProjectDefault());
@@ -203,7 +208,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
final VirtualFile fileA = createFile("A.java");
final VirtualFile fileB = createFile("B.properties");
assertEquals(getProjectDefault(), fileA.getCharset());
assertEquals(StandardCharsets.UTF_8, fileB.getCharset());
assertEquals(UTF_8, fileB.getCharset());
EncodingProjectManager.getInstance(myProject).setEncoding(fileA, WINDOWS_1251);
EncodingProjectManager.getInstance(myProject).setEncoding(fileB, WINDOWS_1252);
@@ -218,7 +223,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
final VirtualFile fileA = createFile("A.java");
final VirtualFile fileB = createFile("B.properties");
assertEquals(getProjectDefault(), fileA.getCharset());
assertEquals(StandardCharsets.ISO_8859_1, fileB.getCharset());
assertEquals(ISO_8859_1, fileB.getCharset());
EncodingProjectManager.getInstance(myProject).setEncoding(fileA, WINDOWS_1251);
EncodingProjectManager.getInstance(myProject).setEncoding(fileB, WINDOWS_1252);
@@ -45,7 +45,7 @@ import java.util.Set;
public final class LoadTextUtil {
private static final Logger LOG = Logger.getInstance(LoadTextUtil.class);
public enum AutoDetectionReason { FROM_BOM, FROM_BYTES }
public enum AutoDetectionReason {FROM_BOM, FROM_BYTES}
private static final int UNLIMITED = -1;
@@ -222,7 +222,7 @@ public final class LoadTextUtil {
* should be {@code this.name().contains(CharsetToolkit.UTF8)} for {@link #getOverriddenCharsetByBOM(byte[], Charset)} to work
*/
SevenBitCharset(@NotNull Charset baseCharset) {
super("IJ__7BIT_"+baseCharset.name(), ArrayUtilRt.EMPTY_STRING_ARRAY);
super("IJ__7BIT_" + baseCharset.name(), ArrayUtilRt.EMPTY_STRING_ARRAY);
myBaseCharset = baseCharset;
}
@@ -309,17 +309,21 @@ public final class LoadTextUtil {
setCharsetAutoDetectionReason(file, AutoDetectionReason.FROM_BOM);
}
//TODO RC: this method could be called with 'partial' content (i.e. length < file.length). In this case charset
// detection is not reliable, and charset shouldn't be updated. Otherwise it could lead to incorrect
// detection of the charset (which is a source of errors by itself), and re-detection of correct charset
// later, with consequent WA and property change notification (see VirtualFileManager.notifyPropertyChanged)
// See IJPL-173099 as an example of incorrect behaviour.
// Unfortunately, the method is called from too many places already, and it is quite hard to validate when
// it is called with partial or not partial content. Maybe it is worth trying to detect (length != file.length())
// and skip .setCharset() then? -- not sure how well it'll go with call-sites there partial content is
// used, but call-site still expects the charset to be updated.
file.setCharset(charset);
//This method could be called with 'partial' content (i.e. length < file.length). In this case charset
//detection is not reliable, because some symbols crucial to charset detection may be outside of partial
//content. So, for partial content, file.charset shouldn't be updated -- because it could lead to incorrect
//detection of the charset (which is a source of errors by itself), and re-detection of correct charset
//later, with consequent WA and property change notification (see VirtualFileManager.notifyPropertyChanged)
//See IJPL-173099 as an example of incorrect behaviour.
//Unfortunately, the method is called from too many places already, and it is quite hard to validate when
//it is called with partial or not partial content. Hence, the current solution is just a heuristic: if
//(length == file.length) we do update file.charset, otherwise we use detected charset only temporarily,
//to decode the current chunk of content, but do NOT store it in file.charset.
//I'm not sure if this will solve the issues like IJPL-173099 completely, but at least performance effect must
//be much more limited.
if (file.getLength() == length ) {
file.setCharset(charset);
}
Charset result = charset;
// optimisation
@@ -383,7 +387,8 @@ public final class LoadTextUtil {
}
CharsetToolkit.GuessedEncoding guessed = toolkit.guessFromContent(0, endOffset);
if (guessed == CharsetToolkit.GuessedEncoding.VALID_UTF8) {
return new DetectResult(StandardCharsets.UTF_8, CharsetToolkit.GuessedEncoding.VALID_UTF8, null); //UTF detected, ignore all directives
return new DetectResult(StandardCharsets.UTF_8, CharsetToolkit.GuessedEncoding.VALID_UTF8,
null); //UTF detected, ignore all directives
}
if (guessed == CharsetToolkit.GuessedEncoding.INVALID_UTF8
&& defaultCharset != StandardCharsets.UTF_8
@@ -484,7 +489,9 @@ public final class LoadTextUtil {
}
}
public static @NotNull Pair.NonNull<Charset, byte[]> chooseMostlyHarmlessCharset(@NotNull Charset existing, @NotNull Charset specified, @NotNull String text) {
public static @NotNull Pair.NonNull<Charset, byte[]> chooseMostlyHarmlessCharset(@NotNull Charset existing,
@NotNull Charset specified,
@NotNull String text) {
try {
if (specified.equals(existing)) {
return Pair.createNonNull(specified, text.getBytes(existing));
@@ -504,11 +511,13 @@ public final class LoadTextUtil {
}
catch (RuntimeException e) {
Charset defaultCharset = Charset.defaultCharset();
return Pair.createNonNull(defaultCharset, text.getBytes(defaultCharset)); //if both are bad and there is no hope, use the default charset
return Pair.createNonNull(defaultCharset,
text.getBytes(defaultCharset)); //if both are bad and there is no hope, use the default charset
}
}
private static byte @Nullable("null means not supported, otherwise it is converted byte stream") [] isSupported(@NotNull Charset charset, @NotNull String str) {
private static byte @Nullable("null means not supported, otherwise it is converted byte stream") [] isSupported(@NotNull Charset charset,
@NotNull String str) {
try {
if (!charset.canEncode()) return null;
byte[] bytes = str.getBytes(charset);
@@ -523,12 +532,16 @@ public final class LoadTextUtil {
}
}
public static @NotNull Charset extractCharsetFromFileContent(@Nullable Project project, @NotNull VirtualFile virtualFile, @NotNull CharSequence text) {
public static @NotNull Charset extractCharsetFromFileContent(@Nullable Project project,
@NotNull VirtualFile virtualFile,
@NotNull CharSequence text) {
Charset value = charsetFromContentOrNull(project, virtualFile, text);
return value == null ? virtualFile.getCharset() : value;
}
public static @Nullable("returns null if cannot determine from content") Charset charsetFromContentOrNull(@Nullable Project project, @NotNull VirtualFile virtualFile, @NotNull CharSequence text) {
public static @Nullable("returns null if cannot determine from content") Charset charsetFromContentOrNull(@Nullable Project project,
@NotNull VirtualFile virtualFile,
@NotNull CharSequence text) {
return CharsetUtil.extractCharsetFromFileContent(project, virtualFile, virtualFile.getFileType(), text);
}
@@ -559,7 +572,8 @@ public final class LoadTextUtil {
public static @NotNull CharSequence loadText(@NotNull VirtualFile file, int limit) {
FileType type = file.getFileType();
if (type.isBinary()) {
throw new IllegalArgumentException("Attempt to load truncated text for binary file: " + file.getPresentableUrl() + ". File type: " + type.getName());
throw new IllegalArgumentException(
"Attempt to load truncated text for binary file: " + file.getPresentableUrl() + ". File type: " + type.getName());
}
if (file instanceof LightVirtualFile) {
@@ -607,7 +621,8 @@ public final class LoadTextUtil {
saveDetectedSeparators, info, info.hardCodedCharset);
if (applyTextTransformer) {
return TextPresentationTransformers.fromPersistent(result.text, virtualFile);
} else {
}
else {
return result.text;
}
}
@@ -639,12 +654,15 @@ public final class LoadTextUtil {
Charset internalCharset = detectResult.hardCodedCharset;
CharsetToolkit.GuessedEncoding guessed = detectResult.guessed;
CharSequence toProcess;
if (internalCharset == null || internalCharset.equals(StandardCharsets.UTF_8) && (guessed == CharsetToolkit.GuessedEncoding.BINARY || guessed == CharsetToolkit.GuessedEncoding.INVALID_UTF8)) {
if (internalCharset == null ||
internalCharset.equals(StandardCharsets.UTF_8) &&
(guessed == CharsetToolkit.GuessedEncoding.BINARY || guessed == CharsetToolkit.GuessedEncoding.INVALID_UTF8)) {
// the charset was not detected, so the file is likely binary
toProcess = null;
}
else {
ConvertResult result = convertBytesAndSetSeparator(buffer, bytes.length(), virtualFile, saveDetectedSeparators, detectResult, internalCharset);
ConvertResult result =
convertBytesAndSetSeparator(buffer, bytes.length(), virtualFile, saveDetectedSeparators, detectResult, internalCharset);
toProcess = result.text;
}
return fileTextProcessor.fun(toProcess);
@@ -683,7 +701,8 @@ public final class LoadTextUtil {
getTextByBinaryPresentation(file.contentsToByteArray(), file);
lineSeparator = file.getDetectedLineSeparator();
}
catch (IOException ignored) { }
catch (IOException ignored) {
}
}
return lineSeparator;
}
@@ -699,7 +718,7 @@ public final class LoadTextUtil {
private static @NotNull ConvertResult convertBytes(byte @NotNull [] bytes,
int startOffset, int endOffset,
@NotNull Charset internalCharset) {
assert startOffset >= 0 && startOffset <= endOffset && endOffset <= bytes.length: startOffset + "," + endOffset+": "+bytes.length;
assert startOffset >= 0 && startOffset <= endOffset && endOffset <= bytes.length : startOffset + "," + endOffset + ": " + bytes.length;
if (internalCharset instanceof SevenBitCharset || internalCharset == StandardCharsets.US_ASCII) {
// optimization: skip byte-to-char conversion for ascii chars
return convertLineSeparatorsToSlashN(bytes, startOffset, endOffset);
@@ -731,7 +750,7 @@ public final class LoadTextUtil {
this.CRLF_count = CRLF_count;
}
String majorLineSeparator () {
String majorLineSeparator() {
String detectedLineSeparator = null;
if (CRLF_count > CR_count && CRLF_count > LF_count) {
detectedLineSeparator = "\r\n";