diff --git a/java/compiler/tests/com/intellij/compiler/CompilerEncodingServiceTest.java b/java/compiler/tests/com/intellij/compiler/CompilerEncodingServiceTest.java index 8f04ad23e9e1..ef7ce20ee233 100644 --- a/java/compiler/tests/com/intellij/compiler/CompilerEncodingServiceTest.java +++ b/java/compiler/tests/com/intellij/compiler/CompilerEncodingServiceTest.java @@ -1,4 +1,4 @@ -// Copyright 2000-2024 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license. +// Copyright 2000-2025 JetBrains s.r.o. and contributors. Use of this source code is governed by the Apache 2.0 license. package com.intellij.compiler; import com.intellij.openapi.application.WriteAction; @@ -15,12 +15,14 @@ import org.jetbrains.annotations.NotNull; import java.io.IOException; import java.nio.charset.Charset; -import java.nio.charset.StandardCharsets; import java.util.Arrays; import java.util.Collection; import java.util.HashSet; import java.util.Set; +import static java.nio.charset.StandardCharsets.ISO_8859_1; +import static java.nio.charset.StandardCharsets.UTF_8; + public class CompilerEncodingServiceTest extends JavaPsiTestCase { private static final Charset WINDOWS_1251 = Charset.forName("windows-1251"); private static final Charset WINDOWS_1252 = Charset.forName("windows-1252"); @@ -68,7 +70,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase { public void testPropertiesEncodingTest() { final VirtualFile file = createFile("A.properties"); - assertEquals(StandardCharsets.UTF_8, file.getCharset()); + assertEquals(UTF_8, file.getCharset()); EncodingProjectManager.getInstance(myProject).setEncoding(file, WINDOWS_1251); assertSameElements(getService().getAllModuleEncodings(myModule), getProjectDefault()); @@ -81,30 +83,30 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase { final VirtualFile file = createFile("test.properties"); WriteAction.run(() -> { content.set(("one=1\n" + - "two=2\n").getBytes(StandardCharsets.ISO_8859_1)); + "two=2\n").getBytes(ISO_8859_1)); file.setBinaryContent(content.get()); }); file.setCharset(null); - assertEquals(StandardCharsets.UTF_8, file.getCharset()); + assertEquals(UTF_8, file.getCharset()); // WriteAction.run(() -> { content.set(ArrayUtil.mergeArrays(content.get(), ("three=3️⃣\n" + - "four=4️⃣\n").getBytes(StandardCharsets.UTF_8))); + "four=4️⃣\n").getBytes(UTF_8))); file.setBinaryContent(content.get()); }); file.setCharset(null); - assertEquals(StandardCharsets.UTF_8, file.getCharset()); + assertEquals(UTF_8, file.getCharset()); // WriteAction.run(() -> { content.set(ArrayUtil.mergeArrays(content.get(), ("five=fünf\n" + - "six=sechs\n").getBytes(StandardCharsets.ISO_8859_1))); + "six=sechs\n").getBytes(ISO_8859_1))); file.setBinaryContent(content.get()); }); file.setCharset(null); - assertEquals(StandardCharsets.ISO_8859_1, file.getCharset()); + assertEquals(ISO_8859_1, file.getCharset()); } public void testBigPropertiesAutoEncoding() throws IOException { @@ -113,15 +115,18 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase { WriteAction.run(() -> { @SuppressWarnings("NonAsciiCharacters") byte[] bytes = "verifyEmail.tooltip=初回ログイン後またはアドレスの変更が送信された後に、ユーザーに自分の電子メールアドレスを確認するように要求します。\n" - .repeat(64).getBytes(StandardCharsets.UTF_8); + .repeat(64).getBytes(UTF_8); file.setBinaryContent(bytes); }); LoadTextUtil.loadText(file, 1024); - assertEquals(StandardCharsets.UTF_8, file.getCharset()); + assertEquals(UTF_8, file.getCharset()); } - public void testCheckEncodingStability() throws IOException { + //TODO RC: Test is temporarily ignored because it relies on LoadTextUtil.loadText(file, length) to always set file + // encoding even by partial content -> nowadays the method .loadText(file, length) only set encoding if file.length==length. + // The test must be updated accordingly, and re-enabled afterwards. + public void _testCheckEncodingStability() throws IOException { final VirtualFile file = createFile("test.properties"); { byte[] bytes = """ @@ -129,7 +134,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase { two=zwei three=drei four=vier - """.getBytes(StandardCharsets.ISO_8859_1); + """.getBytes(ISO_8859_1); WriteAction.run(() -> { file.setBinaryContent(bytes); }); @@ -138,7 +143,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase { CharSequence text = LoadTextUtil.loadText(file, i); assertEquals("Text encoding mismatch: expected UTF-8 the entire file. " + "Found: " + file.getCharset() + " in part of content: '" + text + "'", - StandardCharsets.UTF_8, + UTF_8, file.getCharset()); } } @@ -157,7 +162,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase { eight=acht nine=neun ten=zehn - """.getBytes(StandardCharsets.ISO_8859_1); + """.getBytes(ISO_8859_1); WriteAction.run(() -> { file.setBinaryContent(bytes); }); @@ -166,7 +171,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase { CharSequence text = LoadTextUtil.loadText(file, i); assertEquals("Text encoding mismatch: expected ISO-8859-1 for the entire file. " + "Found: " + file.getCharset() + " in part of content: '" + text + "'", - StandardCharsets.ISO_8859_1, + ISO_8859_1, file.getCharset()); } } @@ -177,7 +182,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase { try { registryValue.setValue(true); final VirtualFile file = createFile("A.properties"); - assertEquals(StandardCharsets.ISO_8859_1, file.getCharset()); + assertEquals(ISO_8859_1, file.getCharset()); EncodingProjectManager.getInstance(myProject).setEncoding(file, WINDOWS_1251); assertSameElements(getService().getAllModuleEncodings(myModule), getProjectDefault()); @@ -203,7 +208,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase { final VirtualFile fileA = createFile("A.java"); final VirtualFile fileB = createFile("B.properties"); assertEquals(getProjectDefault(), fileA.getCharset()); - assertEquals(StandardCharsets.UTF_8, fileB.getCharset()); + assertEquals(UTF_8, fileB.getCharset()); EncodingProjectManager.getInstance(myProject).setEncoding(fileA, WINDOWS_1251); EncodingProjectManager.getInstance(myProject).setEncoding(fileB, WINDOWS_1252); @@ -218,7 +223,7 @@ public class CompilerEncodingServiceTest extends JavaPsiTestCase { final VirtualFile fileA = createFile("A.java"); final VirtualFile fileB = createFile("B.properties"); assertEquals(getProjectDefault(), fileA.getCharset()); - assertEquals(StandardCharsets.ISO_8859_1, fileB.getCharset()); + assertEquals(ISO_8859_1, fileB.getCharset()); EncodingProjectManager.getInstance(myProject).setEncoding(fileA, WINDOWS_1251); EncodingProjectManager.getInstance(myProject).setEncoding(fileB, WINDOWS_1252); diff --git a/platform/core-impl/src/com/intellij/openapi/fileEditor/impl/LoadTextUtil.java b/platform/core-impl/src/com/intellij/openapi/fileEditor/impl/LoadTextUtil.java index d5127bf8201a..0a9fcf87fc4c 100644 --- a/platform/core-impl/src/com/intellij/openapi/fileEditor/impl/LoadTextUtil.java +++ b/platform/core-impl/src/com/intellij/openapi/fileEditor/impl/LoadTextUtil.java @@ -45,7 +45,7 @@ import java.util.Set; public final class LoadTextUtil { private static final Logger LOG = Logger.getInstance(LoadTextUtil.class); - public enum AutoDetectionReason { FROM_BOM, FROM_BYTES } + public enum AutoDetectionReason {FROM_BOM, FROM_BYTES} private static final int UNLIMITED = -1; @@ -222,7 +222,7 @@ public final class LoadTextUtil { * should be {@code this.name().contains(CharsetToolkit.UTF8)} for {@link #getOverriddenCharsetByBOM(byte[], Charset)} to work */ SevenBitCharset(@NotNull Charset baseCharset) { - super("IJ__7BIT_"+baseCharset.name(), ArrayUtilRt.EMPTY_STRING_ARRAY); + super("IJ__7BIT_" + baseCharset.name(), ArrayUtilRt.EMPTY_STRING_ARRAY); myBaseCharset = baseCharset; } @@ -309,17 +309,21 @@ public final class LoadTextUtil { setCharsetAutoDetectionReason(file, AutoDetectionReason.FROM_BOM); } - //TODO RC: this method could be called with 'partial' content (i.e. length < file.length). In this case charset - // detection is not reliable, and charset shouldn't be updated. Otherwise it could lead to incorrect - // detection of the charset (which is a source of errors by itself), and re-detection of correct charset - // later, with consequent WA and property change notification (see VirtualFileManager.notifyPropertyChanged) - // See IJPL-173099 as an example of incorrect behaviour. - // Unfortunately, the method is called from too many places already, and it is quite hard to validate when - // it is called with partial or not partial content. Maybe it is worth trying to detect (length != file.length()) - // and skip .setCharset() then? -- not sure how well it'll go with call-sites there partial content is - // used, but call-site still expects the charset to be updated. - - file.setCharset(charset); + //This method could be called with 'partial' content (i.e. length < file.length). In this case charset + //detection is not reliable, because some symbols crucial to charset detection may be outside of partial + //content. So, for partial content, file.charset shouldn't be updated -- because it could lead to incorrect + //detection of the charset (which is a source of errors by itself), and re-detection of correct charset + //later, with consequent WA and property change notification (see VirtualFileManager.notifyPropertyChanged) + //See IJPL-173099 as an example of incorrect behaviour. + //Unfortunately, the method is called from too many places already, and it is quite hard to validate when + //it is called with partial or not partial content. Hence, the current solution is just a heuristic: if + //(length == file.length) we do update file.charset, otherwise we use detected charset only temporarily, + //to decode the current chunk of content, but do NOT store it in file.charset. + //I'm not sure if this will solve the issues like IJPL-173099 completely, but at least performance effect must + //be much more limited. + if (file.getLength() == length ) { + file.setCharset(charset); + } Charset result = charset; // optimisation @@ -383,7 +387,8 @@ public final class LoadTextUtil { } CharsetToolkit.GuessedEncoding guessed = toolkit.guessFromContent(0, endOffset); if (guessed == CharsetToolkit.GuessedEncoding.VALID_UTF8) { - return new DetectResult(StandardCharsets.UTF_8, CharsetToolkit.GuessedEncoding.VALID_UTF8, null); //UTF detected, ignore all directives + return new DetectResult(StandardCharsets.UTF_8, CharsetToolkit.GuessedEncoding.VALID_UTF8, + null); //UTF detected, ignore all directives } if (guessed == CharsetToolkit.GuessedEncoding.INVALID_UTF8 && defaultCharset != StandardCharsets.UTF_8 @@ -484,7 +489,9 @@ public final class LoadTextUtil { } } - public static @NotNull Pair.NonNull chooseMostlyHarmlessCharset(@NotNull Charset existing, @NotNull Charset specified, @NotNull String text) { + public static @NotNull Pair.NonNull chooseMostlyHarmlessCharset(@NotNull Charset existing, + @NotNull Charset specified, + @NotNull String text) { try { if (specified.equals(existing)) { return Pair.createNonNull(specified, text.getBytes(existing)); @@ -504,11 +511,13 @@ public final class LoadTextUtil { } catch (RuntimeException e) { Charset defaultCharset = Charset.defaultCharset(); - return Pair.createNonNull(defaultCharset, text.getBytes(defaultCharset)); //if both are bad and there is no hope, use the default charset + return Pair.createNonNull(defaultCharset, + text.getBytes(defaultCharset)); //if both are bad and there is no hope, use the default charset } } - private static byte @Nullable("null means not supported, otherwise it is converted byte stream") [] isSupported(@NotNull Charset charset, @NotNull String str) { + private static byte @Nullable("null means not supported, otherwise it is converted byte stream") [] isSupported(@NotNull Charset charset, + @NotNull String str) { try { if (!charset.canEncode()) return null; byte[] bytes = str.getBytes(charset); @@ -523,12 +532,16 @@ public final class LoadTextUtil { } } - public static @NotNull Charset extractCharsetFromFileContent(@Nullable Project project, @NotNull VirtualFile virtualFile, @NotNull CharSequence text) { + public static @NotNull Charset extractCharsetFromFileContent(@Nullable Project project, + @NotNull VirtualFile virtualFile, + @NotNull CharSequence text) { Charset value = charsetFromContentOrNull(project, virtualFile, text); return value == null ? virtualFile.getCharset() : value; } - public static @Nullable("returns null if cannot determine from content") Charset charsetFromContentOrNull(@Nullable Project project, @NotNull VirtualFile virtualFile, @NotNull CharSequence text) { + public static @Nullable("returns null if cannot determine from content") Charset charsetFromContentOrNull(@Nullable Project project, + @NotNull VirtualFile virtualFile, + @NotNull CharSequence text) { return CharsetUtil.extractCharsetFromFileContent(project, virtualFile, virtualFile.getFileType(), text); } @@ -559,7 +572,8 @@ public final class LoadTextUtil { public static @NotNull CharSequence loadText(@NotNull VirtualFile file, int limit) { FileType type = file.getFileType(); if (type.isBinary()) { - throw new IllegalArgumentException("Attempt to load truncated text for binary file: " + file.getPresentableUrl() + ". File type: " + type.getName()); + throw new IllegalArgumentException( + "Attempt to load truncated text for binary file: " + file.getPresentableUrl() + ". File type: " + type.getName()); } if (file instanceof LightVirtualFile) { @@ -607,7 +621,8 @@ public final class LoadTextUtil { saveDetectedSeparators, info, info.hardCodedCharset); if (applyTextTransformer) { return TextPresentationTransformers.fromPersistent(result.text, virtualFile); - } else { + } + else { return result.text; } } @@ -639,12 +654,15 @@ public final class LoadTextUtil { Charset internalCharset = detectResult.hardCodedCharset; CharsetToolkit.GuessedEncoding guessed = detectResult.guessed; CharSequence toProcess; - if (internalCharset == null || internalCharset.equals(StandardCharsets.UTF_8) && (guessed == CharsetToolkit.GuessedEncoding.BINARY || guessed == CharsetToolkit.GuessedEncoding.INVALID_UTF8)) { + if (internalCharset == null || + internalCharset.equals(StandardCharsets.UTF_8) && + (guessed == CharsetToolkit.GuessedEncoding.BINARY || guessed == CharsetToolkit.GuessedEncoding.INVALID_UTF8)) { // the charset was not detected, so the file is likely binary toProcess = null; } else { - ConvertResult result = convertBytesAndSetSeparator(buffer, bytes.length(), virtualFile, saveDetectedSeparators, detectResult, internalCharset); + ConvertResult result = + convertBytesAndSetSeparator(buffer, bytes.length(), virtualFile, saveDetectedSeparators, detectResult, internalCharset); toProcess = result.text; } return fileTextProcessor.fun(toProcess); @@ -683,7 +701,8 @@ public final class LoadTextUtil { getTextByBinaryPresentation(file.contentsToByteArray(), file); lineSeparator = file.getDetectedLineSeparator(); } - catch (IOException ignored) { } + catch (IOException ignored) { + } } return lineSeparator; } @@ -699,7 +718,7 @@ public final class LoadTextUtil { private static @NotNull ConvertResult convertBytes(byte @NotNull [] bytes, int startOffset, int endOffset, @NotNull Charset internalCharset) { - assert startOffset >= 0 && startOffset <= endOffset && endOffset <= bytes.length: startOffset + "," + endOffset+": "+bytes.length; + assert startOffset >= 0 && startOffset <= endOffset && endOffset <= bytes.length : startOffset + "," + endOffset + ": " + bytes.length; if (internalCharset instanceof SevenBitCharset || internalCharset == StandardCharsets.US_ASCII) { // optimization: skip byte-to-char conversion for ascii chars return convertLineSeparatorsToSlashN(bytes, startOffset, endOffset); @@ -731,7 +750,7 @@ public final class LoadTextUtil { this.CRLF_count = CRLF_count; } - String majorLineSeparator () { + String majorLineSeparator() { String detectedLineSeparator = null; if (CRLF_count > CR_count && CRLF_count > LF_count) { detectedLineSeparator = "\r\n";