mirror of
https://gitflic.ru/project/openide/openide.git
synced 2026-09-27 10:03:11 +07:00
[vfs] IJPL-173099, IJPL-191579: avoid detecting charset by partial-content
+ Do not update VirtualFile.charset if only part of it's content bytes are available for charset-detection GitOrigin-RevId: 91fc98cd58ef3539e96ebf051f7bb9317b633289
This commit is contained in:
committed by
intellij-monorepo-bot
parent
798fac4129
commit
e805de8996
@@ -1,4 +1,4 @@
|
||||
// Copyright 2000-2024 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
// Copyright 2000-2025 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license.
|
||||
package com.intellij.compiler;
|
||||
|
||||
import com.intellij.openapi.application.WriteAction;
|
||||
@@ -15,12 +15,14 @@ import org.jetbrains.annotations.NotNull;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.charset.Charset;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.Arrays;
|
||||
import java.util.Collection;
|
||||
import java.util.HashSet;
|
||||
import java.util.Set;
|
||||
|
||||
import static java.nio.charset.StandardCharsets.ISO_8859_1;
|
||||
import static java.nio.charset.StandardCharsets.UTF_8;
|
||||
|
||||
public class CompilerEncodingServiceTest extends JavaPsiTestCase {
|
||||
private static final Charset WINDOWS_1251 = Charset.forName("windows-1251");
|
||||
private static final Charset WINDOWS_1252 = Charset.forName("windows-1252");
|
||||
@@ -68,7 +70,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
|
||||
|
||||
public void testPropertiesEncodingTest() {
|
||||
final VirtualFile file = createFile("A.properties");
|
||||
assertEquals(StandardCharsets.UTF_8, file.getCharset());
|
||||
assertEquals(UTF_8, file.getCharset());
|
||||
EncodingProjectManager.getInstance(myProject).setEncoding(file, WINDOWS_1251);
|
||||
|
||||
assertSameElements(getService().getAllModuleEncodings(myModule), getProjectDefault());
|
||||
@@ -81,30 +83,30 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
|
||||
final VirtualFile file = createFile("test.properties");
|
||||
WriteAction.run(() -> {
|
||||
content.set(("one=1\n" +
|
||||
"two=2\n").getBytes(StandardCharsets.ISO_8859_1));
|
||||
"two=2\n").getBytes(ISO_8859_1));
|
||||
file.setBinaryContent(content.get());
|
||||
});
|
||||
file.setCharset(null);
|
||||
assertEquals(StandardCharsets.UTF_8, file.getCharset());
|
||||
assertEquals(UTF_8, file.getCharset());
|
||||
|
||||
//
|
||||
WriteAction.run(() -> {
|
||||
content.set(ArrayUtil.mergeArrays(content.get(), ("three=3️⃣\n" +
|
||||
"four=4️⃣\n").getBytes(StandardCharsets.UTF_8)));
|
||||
"four=4️⃣\n").getBytes(UTF_8)));
|
||||
file.setBinaryContent(content.get());
|
||||
});
|
||||
file.setCharset(null);
|
||||
assertEquals(StandardCharsets.UTF_8, file.getCharset());
|
||||
assertEquals(UTF_8, file.getCharset());
|
||||
|
||||
//
|
||||
WriteAction.run(() -> {
|
||||
content.set(ArrayUtil.mergeArrays(content.get(), ("five=fünf\n" +
|
||||
"six=sechs\n").getBytes(StandardCharsets.ISO_8859_1)));
|
||||
"six=sechs\n").getBytes(ISO_8859_1)));
|
||||
|
||||
file.setBinaryContent(content.get());
|
||||
});
|
||||
file.setCharset(null);
|
||||
assertEquals(StandardCharsets.ISO_8859_1, file.getCharset());
|
||||
assertEquals(ISO_8859_1, file.getCharset());
|
||||
}
|
||||
|
||||
public void testBigPropertiesAutoEncoding() throws IOException {
|
||||
@@ -113,15 +115,18 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
|
||||
WriteAction.run(() -> {
|
||||
@SuppressWarnings("NonAsciiCharacters")
|
||||
byte[] bytes = "verifyEmail.tooltip=初回ログイン後またはアドレスの変更が送信された後に、ユーザーに自分の電子メールアドレスを確認するように要求します。\n"
|
||||
.repeat(64).getBytes(StandardCharsets.UTF_8);
|
||||
.repeat(64).getBytes(UTF_8);
|
||||
file.setBinaryContent(bytes);
|
||||
});
|
||||
|
||||
LoadTextUtil.loadText(file, 1024);
|
||||
assertEquals(StandardCharsets.UTF_8, file.getCharset());
|
||||
assertEquals(UTF_8, file.getCharset());
|
||||
}
|
||||
|
||||
public void testCheckEncodingStability() throws IOException {
|
||||
//TODO RC: Test is temporarily ignored because it relies on LoadTextUtil.loadText(file, length) to always set file
|
||||
// encoding even by partial content -> nowadays the method .loadText(file, length) only set encoding if file.length==length.
|
||||
// The test must be updated accordingly, and re-enabled afterwards.
|
||||
public void _testCheckEncodingStability() throws IOException {
|
||||
final VirtualFile file = createFile("test.properties");
|
||||
{
|
||||
byte[] bytes = """
|
||||
@@ -129,7 +134,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
|
||||
two=zwei
|
||||
three=drei
|
||||
four=vier
|
||||
""".getBytes(StandardCharsets.ISO_8859_1);
|
||||
""".getBytes(ISO_8859_1);
|
||||
WriteAction.run(() -> {
|
||||
file.setBinaryContent(bytes);
|
||||
});
|
||||
@@ -138,7 +143,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
|
||||
CharSequence text = LoadTextUtil.loadText(file, i);
|
||||
assertEquals("Text encoding mismatch: expected UTF-8 the entire file. " +
|
||||
"Found: " + file.getCharset() + " in part of content: '" + text + "'",
|
||||
StandardCharsets.UTF_8,
|
||||
UTF_8,
|
||||
file.getCharset());
|
||||
}
|
||||
}
|
||||
@@ -157,7 +162,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
|
||||
eight=acht
|
||||
nine=neun
|
||||
ten=zehn
|
||||
""".getBytes(StandardCharsets.ISO_8859_1);
|
||||
""".getBytes(ISO_8859_1);
|
||||
WriteAction.run(() -> {
|
||||
file.setBinaryContent(bytes);
|
||||
});
|
||||
@@ -166,7 +171,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
|
||||
CharSequence text = LoadTextUtil.loadText(file, i);
|
||||
assertEquals("Text encoding mismatch: expected ISO-8859-1 for the entire file. " +
|
||||
"Found: " + file.getCharset() + " in part of content: '" + text + "'",
|
||||
StandardCharsets.ISO_8859_1,
|
||||
ISO_8859_1,
|
||||
file.getCharset());
|
||||
}
|
||||
}
|
||||
@@ -177,7 +182,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
|
||||
try {
|
||||
registryValue.setValue(true);
|
||||
final VirtualFile file = createFile("A.properties");
|
||||
assertEquals(StandardCharsets.ISO_8859_1, file.getCharset());
|
||||
assertEquals(ISO_8859_1, file.getCharset());
|
||||
EncodingProjectManager.getInstance(myProject).setEncoding(file, WINDOWS_1251);
|
||||
|
||||
assertSameElements(getService().getAllModuleEncodings(myModule), getProjectDefault());
|
||||
@@ -203,7 +208,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
|
||||
final VirtualFile fileA = createFile("A.java");
|
||||
final VirtualFile fileB = createFile("B.properties");
|
||||
assertEquals(getProjectDefault(), fileA.getCharset());
|
||||
assertEquals(StandardCharsets.UTF_8, fileB.getCharset());
|
||||
assertEquals(UTF_8, fileB.getCharset());
|
||||
EncodingProjectManager.getInstance(myProject).setEncoding(fileA, WINDOWS_1251);
|
||||
EncodingProjectManager.getInstance(myProject).setEncoding(fileB, WINDOWS_1252);
|
||||
|
||||
@@ -218,7 +223,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase {
|
||||
final VirtualFile fileA = createFile("A.java");
|
||||
final VirtualFile fileB = createFile("B.properties");
|
||||
assertEquals(getProjectDefault(), fileA.getCharset());
|
||||
assertEquals(StandardCharsets.ISO_8859_1, fileB.getCharset());
|
||||
assertEquals(ISO_8859_1, fileB.getCharset());
|
||||
EncodingProjectManager.getInstance(myProject).setEncoding(fileA, WINDOWS_1251);
|
||||
EncodingProjectManager.getInstance(myProject).setEncoding(fileB, WINDOWS_1252);
|
||||
|
||||
|
||||
@@ -45,7 +45,7 @@ import java.util.Set;
|
||||
public final class LoadTextUtil {
|
||||
private static final Logger LOG = Logger.getInstance(LoadTextUtil.class);
|
||||
|
||||
public enum AutoDetectionReason { FROM_BOM, FROM_BYTES }
|
||||
public enum AutoDetectionReason {FROM_BOM, FROM_BYTES}
|
||||
|
||||
private static final int UNLIMITED = -1;
|
||||
|
||||
@@ -222,7 +222,7 @@ public final class LoadTextUtil {
|
||||
* should be {@code this.name().contains(CharsetToolkit.UTF8)} for {@link #getOverriddenCharsetByBOM(byte[], Charset)} to work
|
||||
*/
|
||||
SevenBitCharset(@NotNull Charset baseCharset) {
|
||||
super("IJ__7BIT_"+baseCharset.name(), ArrayUtilRt.EMPTY_STRING_ARRAY);
|
||||
super("IJ__7BIT_" + baseCharset.name(), ArrayUtilRt.EMPTY_STRING_ARRAY);
|
||||
myBaseCharset = baseCharset;
|
||||
}
|
||||
|
||||
@@ -309,17 +309,21 @@ public final class LoadTextUtil {
|
||||
setCharsetAutoDetectionReason(file, AutoDetectionReason.FROM_BOM);
|
||||
}
|
||||
|
||||
//TODO RC: this method could be called with 'partial' content (i.e. length < file.length). In this case charset
|
||||
// detection is not reliable, and charset shouldn't be updated. Otherwise it could lead to incorrect
|
||||
// detection of the charset (which is a source of errors by itself), and re-detection of correct charset
|
||||
// later, with consequent WA and property change notification (see VirtualFileManager.notifyPropertyChanged)
|
||||
// See IJPL-173099 as an example of incorrect behaviour.
|
||||
// Unfortunately, the method is called from too many places already, and it is quite hard to validate when
|
||||
// it is called with partial or not partial content. Maybe it is worth trying to detect (length != file.length())
|
||||
// and skip .setCharset() then? -- not sure how well it'll go with call-sites there partial content is
|
||||
// used, but call-site still expects the charset to be updated.
|
||||
|
||||
file.setCharset(charset);
|
||||
//This method could be called with 'partial' content (i.e. length < file.length). In this case charset
|
||||
//detection is not reliable, because some symbols crucial to charset detection may be outside of partial
|
||||
//content. So, for partial content, file.charset shouldn't be updated -- because it could lead to incorrect
|
||||
//detection of the charset (which is a source of errors by itself), and re-detection of correct charset
|
||||
//later, with consequent WA and property change notification (see VirtualFileManager.notifyPropertyChanged)
|
||||
//See IJPL-173099 as an example of incorrect behaviour.
|
||||
//Unfortunately, the method is called from too many places already, and it is quite hard to validate when
|
||||
//it is called with partial or not partial content. Hence, the current solution is just a heuristic: if
|
||||
//(length == file.length) we do update file.charset, otherwise we use detected charset only temporarily,
|
||||
//to decode the current chunk of content, but do NOT store it in file.charset.
|
||||
//I'm not sure if this will solve the issues like IJPL-173099 completely, but at least performance effect must
|
||||
//be much more limited.
|
||||
if (file.getLength() == length ) {
|
||||
file.setCharset(charset);
|
||||
}
|
||||
|
||||
Charset result = charset;
|
||||
// optimisation
|
||||
@@ -383,7 +387,8 @@ public final class LoadTextUtil {
|
||||
}
|
||||
CharsetToolkit.GuessedEncoding guessed = toolkit.guessFromContent(0, endOffset);
|
||||
if (guessed == CharsetToolkit.GuessedEncoding.VALID_UTF8) {
|
||||
return new DetectResult(StandardCharsets.UTF_8, CharsetToolkit.GuessedEncoding.VALID_UTF8, null); //UTF detected, ignore all directives
|
||||
return new DetectResult(StandardCharsets.UTF_8, CharsetToolkit.GuessedEncoding.VALID_UTF8,
|
||||
null); //UTF detected, ignore all directives
|
||||
}
|
||||
if (guessed == CharsetToolkit.GuessedEncoding.INVALID_UTF8
|
||||
&& defaultCharset != StandardCharsets.UTF_8
|
||||
@@ -484,7 +489,9 @@ public final class LoadTextUtil {
|
||||
}
|
||||
}
|
||||
|
||||
public static @NotNull Pair.NonNull<Charset, byte[]> chooseMostlyHarmlessCharset(@NotNull Charset existing, @NotNull Charset specified, @NotNull String text) {
|
||||
public static @NotNull Pair.NonNull<Charset, byte[]> chooseMostlyHarmlessCharset(@NotNull Charset existing,
|
||||
@NotNull Charset specified,
|
||||
@NotNull String text) {
|
||||
try {
|
||||
if (specified.equals(existing)) {
|
||||
return Pair.createNonNull(specified, text.getBytes(existing));
|
||||
@@ -504,11 +511,13 @@ public final class LoadTextUtil {
|
||||
}
|
||||
catch (RuntimeException e) {
|
||||
Charset defaultCharset = Charset.defaultCharset();
|
||||
return Pair.createNonNull(defaultCharset, text.getBytes(defaultCharset)); //if both are bad and there is no hope, use the default charset
|
||||
return Pair.createNonNull(defaultCharset,
|
||||
text.getBytes(defaultCharset)); //if both are bad and there is no hope, use the default charset
|
||||
}
|
||||
}
|
||||
|
||||
private static byte @Nullable("null means not supported, otherwise it is converted byte stream") [] isSupported(@NotNull Charset charset, @NotNull String str) {
|
||||
private static byte @Nullable("null means not supported, otherwise it is converted byte stream") [] isSupported(@NotNull Charset charset,
|
||||
@NotNull String str) {
|
||||
try {
|
||||
if (!charset.canEncode()) return null;
|
||||
byte[] bytes = str.getBytes(charset);
|
||||
@@ -523,12 +532,16 @@ public final class LoadTextUtil {
|
||||
}
|
||||
}
|
||||
|
||||
public static @NotNull Charset extractCharsetFromFileContent(@Nullable Project project, @NotNull VirtualFile virtualFile, @NotNull CharSequence text) {
|
||||
public static @NotNull Charset extractCharsetFromFileContent(@Nullable Project project,
|
||||
@NotNull VirtualFile virtualFile,
|
||||
@NotNull CharSequence text) {
|
||||
Charset value = charsetFromContentOrNull(project, virtualFile, text);
|
||||
return value == null ? virtualFile.getCharset() : value;
|
||||
}
|
||||
|
||||
public static @Nullable("returns null if cannot determine from content") Charset charsetFromContentOrNull(@Nullable Project project, @NotNull VirtualFile virtualFile, @NotNull CharSequence text) {
|
||||
public static @Nullable("returns null if cannot determine from content") Charset charsetFromContentOrNull(@Nullable Project project,
|
||||
@NotNull VirtualFile virtualFile,
|
||||
@NotNull CharSequence text) {
|
||||
return CharsetUtil.extractCharsetFromFileContent(project, virtualFile, virtualFile.getFileType(), text);
|
||||
}
|
||||
|
||||
@@ -559,7 +572,8 @@ public final class LoadTextUtil {
|
||||
public static @NotNull CharSequence loadText(@NotNull VirtualFile file, int limit) {
|
||||
FileType type = file.getFileType();
|
||||
if (type.isBinary()) {
|
||||
throw new IllegalArgumentException("Attempt to load truncated text for binary file: " + file.getPresentableUrl() + ". File type: " + type.getName());
|
||||
throw new IllegalArgumentException(
|
||||
"Attempt to load truncated text for binary file: " + file.getPresentableUrl() + ". File type: " + type.getName());
|
||||
}
|
||||
|
||||
if (file instanceof LightVirtualFile) {
|
||||
@@ -607,7 +621,8 @@ public final class LoadTextUtil {
|
||||
saveDetectedSeparators, info, info.hardCodedCharset);
|
||||
if (applyTextTransformer) {
|
||||
return TextPresentationTransformers.fromPersistent(result.text, virtualFile);
|
||||
} else {
|
||||
}
|
||||
else {
|
||||
return result.text;
|
||||
}
|
||||
}
|
||||
@@ -639,12 +654,15 @@ public final class LoadTextUtil {
|
||||
Charset internalCharset = detectResult.hardCodedCharset;
|
||||
CharsetToolkit.GuessedEncoding guessed = detectResult.guessed;
|
||||
CharSequence toProcess;
|
||||
if (internalCharset == null || internalCharset.equals(StandardCharsets.UTF_8) && (guessed == CharsetToolkit.GuessedEncoding.BINARY || guessed == CharsetToolkit.GuessedEncoding.INVALID_UTF8)) {
|
||||
if (internalCharset == null ||
|
||||
internalCharset.equals(StandardCharsets.UTF_8) &&
|
||||
(guessed == CharsetToolkit.GuessedEncoding.BINARY || guessed == CharsetToolkit.GuessedEncoding.INVALID_UTF8)) {
|
||||
// the charset was not detected, so the file is likely binary
|
||||
toProcess = null;
|
||||
}
|
||||
else {
|
||||
ConvertResult result = convertBytesAndSetSeparator(buffer, bytes.length(), virtualFile, saveDetectedSeparators, detectResult, internalCharset);
|
||||
ConvertResult result =
|
||||
convertBytesAndSetSeparator(buffer, bytes.length(), virtualFile, saveDetectedSeparators, detectResult, internalCharset);
|
||||
toProcess = result.text;
|
||||
}
|
||||
return fileTextProcessor.fun(toProcess);
|
||||
@@ -683,7 +701,8 @@ public final class LoadTextUtil {
|
||||
getTextByBinaryPresentation(file.contentsToByteArray(), file);
|
||||
lineSeparator = file.getDetectedLineSeparator();
|
||||
}
|
||||
catch (IOException ignored) { }
|
||||
catch (IOException ignored) {
|
||||
}
|
||||
}
|
||||
return lineSeparator;
|
||||
}
|
||||
@@ -699,7 +718,7 @@ public final class LoadTextUtil {
|
||||
private static @NotNull ConvertResult convertBytes(byte @NotNull [] bytes,
|
||||
int startOffset, int endOffset,
|
||||
@NotNull Charset internalCharset) {
|
||||
assert startOffset >= 0 && startOffset <= endOffset && endOffset <= bytes.length: startOffset + "," + endOffset+": "+bytes.length;
|
||||
assert startOffset >= 0 && startOffset <= endOffset && endOffset <= bytes.length : startOffset + "," + endOffset + ": " + bytes.length;
|
||||
if (internalCharset instanceof SevenBitCharset || internalCharset == StandardCharsets.US_ASCII) {
|
||||
// optimization: skip byte-to-char conversion for ascii chars
|
||||
return convertLineSeparatorsToSlashN(bytes, startOffset, endOffset);
|
||||
@@ -731,7 +750,7 @@ public final class LoadTextUtil {
|
||||
this.CRLF_count = CRLF_count;
|
||||
}
|
||||
|
||||
String majorLineSeparator () {
|
||||
String majorLineSeparator() {
|
||||
String detectedLineSeparator = null;
|
||||
if (CRLF_count > CR_count && CRLF_count > LF_count) {
|
||||
detectedLineSeparator = "\r\n";
|
||||
|
||||
Reference in New Issue
Block a user