From 655b408b1116ab9f037d7dc0429dee3e4f58bbbf Mon Sep 17 00:00:00 2001 From: Peter Gromov Date: Mon, 8 Jun 2020 13:52:33 +0200 Subject: [PATCH] extract a more public interface from TokenSequence, use it where possible GitOrigin-RevId: badbfcfc9b39e165f312ee0f01ff155dc242a75b --- .../java/JavaBinaryPlusExpressionIndex.java | 4 +- .../search/JavaNullMethodArgumentIndex.java | 8 +- .../lang/java/parser/JavaParserUtil.java | 3 +- .../psi/impl/source/JavaLightStubBuilder.java | 3 +- .../src/com/intellij/lexer/TokenizedText.java | 192 ++++++++++++++++++ .../com/intellij/lang/impl/TokenSequence.java | 122 ++--------- 6 files changed, 216 insertions(+), 116 deletions(-) create mode 100644 platform/core-api/src/com/intellij/lexer/TokenizedText.java diff --git a/java/java-indexing-impl/src/com/intellij/psi/impl/java/JavaBinaryPlusExpressionIndex.java b/java/java-indexing-impl/src/com/intellij/psi/impl/java/JavaBinaryPlusExpressionIndex.java index cd137a8fad1a..9fe962221f30 100644 --- a/java/java-indexing-impl/src/com/intellij/psi/impl/java/JavaBinaryPlusExpressionIndex.java +++ b/java/java-indexing-impl/src/com/intellij/psi/impl/java/JavaBinaryPlusExpressionIndex.java @@ -2,8 +2,8 @@ package com.intellij.psi.impl.java; import com.intellij.ide.highlighter.JavaFileType; -import com.intellij.lang.impl.TokenSequence; import com.intellij.lang.java.parser.JavaParserUtil; +import com.intellij.lexer.TokenizedText; import com.intellij.openapi.vfs.VirtualFile; import com.intellij.psi.impl.source.JavaFileElementType; import com.intellij.psi.impl.source.tree.ElementType; @@ -37,7 +37,7 @@ public class JavaBinaryPlusExpressionIndex extends FileBasedIndexExtension getIndexer() { return inputData -> { - TokenSequence tokens = JavaParserUtil.obtainTokens(inputData.getPsiFile()); + TokenizedText tokens = JavaParserUtil.obtainTokens(inputData.getPsiFile()); TIntArrayList result = new TIntArrayList(); for (int i = 0; i < tokens.getTokenCount(); i++) { diff --git a/java/java-indexing-impl/src/com/intellij/psi/impl/search/JavaNullMethodArgumentIndex.java b/java/java-indexing-impl/src/com/intellij/psi/impl/search/JavaNullMethodArgumentIndex.java index 0fc72842ef7e..120bead651c7 100644 --- a/java/java-indexing-impl/src/com/intellij/psi/impl/search/JavaNullMethodArgumentIndex.java +++ b/java/java-indexing-impl/src/com/intellij/psi/impl/search/JavaNullMethodArgumentIndex.java @@ -2,9 +2,9 @@ package com.intellij.psi.impl.search; import com.intellij.ide.highlighter.JavaFileType; -import com.intellij.lang.impl.TokenSequence; import com.intellij.lang.java.JavaParserDefinition; import com.intellij.lang.java.parser.JavaParserUtil; +import com.intellij.lexer.TokenizedText; import com.intellij.openapi.application.ApplicationManager; import com.intellij.openapi.vfs.VirtualFile; import com.intellij.psi.tree.IElementType; @@ -46,7 +46,7 @@ public final class JavaNullMethodArgumentIndex extends ScalarIndexExtension result = new THashMap<>(); - TokenSequence tokens = JavaParserUtil.obtainTokens(inputData.getPsiFile()); + TokenizedText tokens = JavaParserUtil.obtainTokens(inputData.getPsiFile()); for (int i = 0; i < tokens.getTokenCount(); i++) { if (tokens.hasType(i, NULL_KEYWORD)) { MethodCallData data = findCallData(tokens, i); @@ -60,7 +60,7 @@ public final class JavaNullMethodArgumentIndex extends ScalarIndexExtension DEEP_PARSE_BLOCKS_IN_STATEMENTS = Key.create("JavaParserUtil.ParserExtender"); @NotNull - public static TokenSequence obtainTokens(@NotNull PsiFile file) { + public static TokenizedText obtainTokens(@NotNull PsiFile file) { return CachedValuesManager.getCachedValue(file, () -> CachedValueProvider.Result.create( TokenSequence.performLexing(file.getViewProvider().getContents(), JavaParserDefinition.createLexer(PsiUtil.getLanguageLevel(file))), diff --git a/java/java-psi-impl/src/com/intellij/psi/impl/source/JavaLightStubBuilder.java b/java/java-psi-impl/src/com/intellij/psi/impl/source/JavaLightStubBuilder.java index 55c7ffea2147..f64b8782810f 100644 --- a/java/java-psi-impl/src/com/intellij/psi/impl/source/JavaLightStubBuilder.java +++ b/java/java-psi-impl/src/com/intellij/psi/impl/source/JavaLightStubBuilder.java @@ -22,6 +22,7 @@ import com.intellij.lang.LighterLazyParseableNode; import com.intellij.lang.impl.TokenSequence; import com.intellij.lang.java.JavaParserDefinition; import com.intellij.lexer.Lexer; +import com.intellij.lexer.TokenizedText; import com.intellij.psi.JavaTokenType; import com.intellij.psi.PsiFile; import com.intellij.psi.PsiJavaFile; @@ -69,7 +70,7 @@ public class JavaLightStubBuilder extends LightStubBuilder { CodeBlockVisitor visitor = new CodeBlockVisitor(); if (TreeUtil.isCollapsedChameleon(node)) { Lexer lexer = JavaParserDefinition.createLexer(PsiUtil.getLanguageLevel(node.getPsi())); - TokenSequence tokens = TokenSequence.performLexing(node.getChars(), lexer); + TokenizedText tokens = TokenSequence.performLexing(node.getChars(), lexer); for (int i = 0; i < tokens.getTokenCount(); i++) { visitor.visit(tokens.getTokenType(i)); } diff --git a/platform/core-api/src/com/intellij/lexer/TokenizedText.java b/platform/core-api/src/com/intellij/lexer/TokenizedText.java new file mode 100644 index 000000000000..abafd42e4480 --- /dev/null +++ b/platform/core-api/src/com/intellij/lexer/TokenizedText.java @@ -0,0 +1,192 @@ +// Copyright 2000-2020 JetBrains s.r.o. Use of this source code is governed by the Apache 2.0 license that can be found in the LICENSE file. +package com.intellij.lexer; + +import com.intellij.openapi.util.Comparing; +import com.intellij.psi.tree.IElementType; +import com.intellij.psi.tree.TokenSet; +import com.intellij.util.ArrayUtil; +import org.jetbrains.annotations.ApiStatus; +import org.jetbrains.annotations.NotNull; +import org.jetbrains.annotations.Nullable; + +/** + * This class represents the result of lexing: text and the tokens produced from it by some lexer. + * It allows clients to inspect all tokens at once and easily move back and forward to implement some simple lexer-based checks. + */ +@ApiStatus.Experimental +public interface TokenizedText { + + /** + * @return the number of tokens inside + */ + int getTokenCount(); + + /** + * @return the full text that was split into the tokens represented here + */ + @NotNull + CharSequence getText(); + + /** + * @return the start offset of the token with the given index + */ + int getTokenStart(int index); + + /** + * @return the end offset of the token with the given index + */ + int getTokenEnd(int index); + + /** + * @return the type of the token with the given index, or null if the index is negative or exceeds token count + */ + IElementType getTokenType(int index); + + /** + * @return the text of the token with the given index, or null if the index is negative or exceeds token count + */ + default CharSequence getTokenText(int index) { + if (index < 0 || index >= getTokenCount()) return null; + return getText().subSequence(getTokenStart(index), getTokenEnd(index)); + } + + /** + * @return whether {@link #getTokenType}(index) would return the given type + */ + default boolean hasType(int index, @NotNull IElementType type) { + return getTokenType(index) == type; + } + + /** + * @return whether {@link #getTokenType}(index) would return any of the given types (null acceptable, indicating start or end of the text) + */ + default boolean hasType(int index, @Nullable IElementType @NotNull ... types) { + return ArrayUtil.contains(getTokenType(index), types); + } + + /** + * @return whether {@link #getTokenType}(index) would return a type in the given set + */ + default boolean hasType(int index, @NotNull TokenSet types) { + return types.contains(getTokenType(index)); + } + + /** + * Moves back, potentially skipping tokens which represent a valid nesting sequence + * with the given types for opening and closing braces. + * @return an index {@code prev} of a token before {@code index} such that either: + *
    + *
  1. {@code prev == index - 1}
  2. + *
  3. {@code hasType(prev + 1, opening) && hasType(index, closing)} and every opening brace between those indices has its closing one before {@code index}
  4. + *
+ */ + default int backWithBraceMatching(int index, @NotNull IElementType opening, @NotNull IElementType closing) { + if (getTokenType(index) == closing) { + int nesting = 1; + while (nesting > 0 && index > 0) { + index--; + IElementType type = getTokenType(index); + if (type == closing) { + nesting++; + } + else if (type == opening) { + nesting--; + } + } + } + return index - 1; + } + + /** + * Moves back from {@code index} while tokens belong to the given set + * @return the largest {@code prev <= index} whose token type doesn't belong to {@code toSkip} + */ + default int backWhile(int index, @NotNull TokenSet toSkip) { + while (hasType(index, toSkip)) { + index--; + } + return index; + } + + /** + * Moves forward from {@code index} while tokens belong to the given set + * @return the smallest {@code next >= index} whose token type doesn't belong to {@code toSkip} + */ + default int forwardWhile(int index, @NotNull TokenSet toSkip) { + while (hasType(index, toSkip)) { + index++; + } + return index; + } + + /** + * @return a view of this object as a {@link Lexer}. + * Note that the returned lexer isn't the same as the one that produced this tokenized text: it returns the same offsets and types, + * but states and positions might differ. The returned lexer may be used to avoid tokenizing the same text again in APIs where lexer is expected, + * but it will only accept the very same text from the very beginning; it can't be used on any other strings. + */ + default @NotNull Lexer asLexer() { + return new WrappingLexer(this); + } + + /** + * A simple lexer over {@link TokenizedText}. + */ + class WrappingLexer extends LexerBase { + private final TokenizedText myTokens; + private int myIndex; + + WrappingLexer(TokenizedText tokens) { + this.myTokens = tokens; + } + + public TokenizedText getTokens() { + return myTokens; + } + + @Override + public void start(@NotNull CharSequence buffer, int startOffset, int endOffset, int initialState) { + assert Comparing.equal(buffer, myTokens.getText()); + assert startOffset == 0; + assert endOffset == buffer.length(); + assert initialState == 0; + myIndex = 0; + } + + @Override + public int getState() { + return myIndex; + } + + @Override + public @Nullable IElementType getTokenType() { + return myTokens.getTokenType(myIndex); + } + + @Override + public int getTokenStart() { + return myTokens.getTokenStart(myIndex); + } + + @Override + public int getTokenEnd() { + return myTokens.getTokenEnd(myIndex); + } + + @Override + public void advance() { + myIndex++; + } + + @Override + public @NotNull CharSequence getBufferSequence() { + return myTokens.getText(); + } + + @Override + public int getBufferEnd() { + return myTokens.getText().length(); + } + } + +} diff --git a/platform/core-impl/src/com/intellij/lang/impl/TokenSequence.java b/platform/core-impl/src/com/intellij/lang/impl/TokenSequence.java index 9534b38dbe45..f4fef9a17954 100644 --- a/platform/core-impl/src/com/intellij/lang/impl/TokenSequence.java +++ b/platform/core-impl/src/com/intellij/lang/impl/TokenSequence.java @@ -16,19 +16,18 @@ package com.intellij.lang.impl; import com.intellij.lexer.Lexer; -import com.intellij.lexer.LexerBase; +import com.intellij.lexer.TokenizedText; import com.intellij.openapi.diagnostic.Logger; import com.intellij.openapi.progress.ProgressIndicatorProvider; import com.intellij.openapi.util.Comparing; import com.intellij.psi.tree.IElementType; -import com.intellij.psi.tree.TokenSet; import com.intellij.util.ArrayUtil; import org.jetbrains.annotations.ApiStatus; import org.jetbrains.annotations.NotNull; import org.jetbrains.annotations.Nullable; @ApiStatus.Experimental -public class TokenSequence { +public class TokenSequence implements TokenizedText { private static final Logger LOG = Logger.getInstance(TokenSequence.class); private final CharSequence myText; @@ -58,134 +57,41 @@ public class TokenSequence { @NotNull public static TokenSequence performLexing(@NotNull CharSequence text, @NotNull Lexer lexer) { if (lexer instanceof WrappingLexer) { - TokenSequence existing = ((WrappingLexer)lexer).mySequence; - if (Comparing.equal(text, existing.myText)) { + TokenizedText existing = ((WrappingLexer)lexer).getTokens(); + if (existing instanceof TokenSequence && Comparing.equal(text, ((TokenSequence)existing).myText)) { // prevent clients like PsiBuilder from modifying shared token types - return new TokenSequence(existing.lexStarts, existing.lexTypes.clone(), existing.lexemeCount, text); + return new TokenSequence(((TokenSequence)existing).lexStarts, + ((TokenSequence)existing).lexTypes.clone(), + ((TokenSequence)existing).lexemeCount, text); } } return new Builder(text, lexer).performLexing(); } + @Override public int getTokenCount() { return lexemeCount; } + @Override public @Nullable IElementType getTokenType(int index) { if (index < 0 || index >= getTokenCount()) return null; return lexTypes[index]; } - public boolean hasType(int index, @NotNull IElementType type) { - return getTokenType(index) == type; - } - - public boolean hasType(int index, @Nullable IElementType @NotNull ... types) { - return ArrayUtil.contains(getTokenType(index), types); - } - - public boolean hasType(int index, @NotNull TokenSet types) { - return types.contains(getTokenType(index)); - } - - public int backWithBraceMatching(int index, @NotNull IElementType opening, @NotNull IElementType closing) { - if (getTokenType(index) == closing) { - int nesting = 1; - while (nesting > 0 && index > 0) { - index--; - IElementType type = getTokenType(index); - if (type == closing) { - nesting++; - } - else if (type == opening) { - nesting--; - } - } - } - return index - 1; - } - - public int backWhile(int index, @NotNull TokenSet toSkip) { - while (hasType(index, toSkip)) { - index--; - } - return index; - } - - public int forwardWhile(int index, @NotNull TokenSet toSkip) { - while (hasType(index, toSkip)) { - index++; - } - return index; - } - + @Override public int getTokenStart(int index) { return lexStarts[index]; } + @Override public int getTokenEnd(int index) { return lexStarts[index + 1]; } - public CharSequence getTokenText(int index) { - return myText.subSequence(getTokenStart(index), getTokenEnd(index)); - } - - public @NotNull Lexer asLexer() { - return new WrappingLexer(this); - } - - private static class WrappingLexer extends LexerBase { - final TokenSequence mySequence; - private int myIndex; - - WrappingLexer(TokenSequence sequence) { - this.mySequence = sequence; - } - - @Override - public void start(@NotNull CharSequence buffer, int startOffset, int endOffset, int initialState) { - assert buffer.equals(mySequence.myText); - assert startOffset == 0; - assert endOffset == buffer.length(); - assert initialState == 0; - myIndex = 0; - } - - @Override - public int getState() { - return myIndex; - } - - @Override - public @Nullable IElementType getTokenType() { - return mySequence.lexTypes[myIndex]; - } - - @Override - public int getTokenStart() { - return mySequence.lexStarts[myIndex]; - } - - @Override - public int getTokenEnd() { - return mySequence.lexStarts[myIndex + 1]; - } - - @Override - public void advance() { - myIndex++; - } - - @Override - public @NotNull CharSequence getBufferSequence() { - return mySequence.myText; - } - - @Override - public int getBufferEnd() { - return mySequence.myText.length(); - } + @Override + public @NotNull CharSequence getText() { + return myText; } private static class Builder {