extract a more public interface from TokenSequence, use it where possible

GitOrigin-RevId: badbfcfc9b39e165f312ee0f01ff155dc242a75b
This commit is contained in:
Peter Gromov
2020-06-08 14:53:56 +03:00
committed by intellij-monorepo-bot
parent 995e376cbe
commit 655b408b11
6 changed files with 216 additions and 116 deletions
@@ -2,8 +2,8 @@
package com.intellij.psi.impl.java;
import com.intellij.ide.highlighter.JavaFileType;
import com.intellij.lang.impl.TokenSequence;
import com.intellij.lang.java.parser.JavaParserUtil;
import com.intellij.lexer.TokenizedText;
import com.intellij.openapi.vfs.VirtualFile;
import com.intellij.psi.impl.source.JavaFileElementType;
import com.intellij.psi.impl.source.tree.ElementType;
@@ -37,7 +37,7 @@ public class JavaBinaryPlusExpressionIndex extends FileBasedIndexExtension<Boole
@Override
public DataIndexer<Boolean, PlusOffsets, FileContent> getIndexer() {
return inputData -> {
TokenSequence tokens = JavaParserUtil.obtainTokens(inputData.getPsiFile());
TokenizedText tokens = JavaParserUtil.obtainTokens(inputData.getPsiFile());
TIntArrayList result = new TIntArrayList();
for (int i = 0; i < tokens.getTokenCount(); i++) {
@@ -2,9 +2,9 @@
package com.intellij.psi.impl.search;
import com.intellij.ide.highlighter.JavaFileType;
import com.intellij.lang.impl.TokenSequence;
import com.intellij.lang.java.JavaParserDefinition;
import com.intellij.lang.java.parser.JavaParserUtil;
import com.intellij.lexer.TokenizedText;
import com.intellij.openapi.application.ApplicationManager;
import com.intellij.openapi.vfs.VirtualFile;
import com.intellij.psi.tree.IElementType;
@@ -46,7 +46,7 @@ public final class JavaNullMethodArgumentIndex extends ScalarIndexExtension<Java
Map<MethodCallData, Void> result = new THashMap<>();
TokenSequence tokens = JavaParserUtil.obtainTokens(inputData.getPsiFile());
TokenizedText tokens = JavaParserUtil.obtainTokens(inputData.getPsiFile());
for (int i = 0; i < tokens.getTokenCount(); i++) {
if (tokens.hasType(i, NULL_KEYWORD)) {
MethodCallData data = findCallData(tokens, i);
@@ -60,7 +60,7 @@ public final class JavaNullMethodArgumentIndex extends ScalarIndexExtension<Java
}
@Nullable
private static MethodCallData findCallData(TokenSequence tokens, int nullIndex) {
private static MethodCallData findCallData(TokenizedText tokens, int nullIndex) {
if (!tokens.hasType(tokens.forwardWhile(nullIndex + 1, JavaParserUtil.WS_COMMENTS), RPARENTH, COMMA)) return null;
int i = tokens.backWhile(nullIndex - 1, JavaParserUtil.WS_COMMENTS);
@@ -87,7 +87,7 @@ public final class JavaNullMethodArgumentIndex extends ScalarIndexExtension<Java
}
@Nullable
private static String findMethodName(TokenSequence tokens, int lparenth) {
private static String findMethodName(TokenizedText tokens, int lparenth) {
int i = tokens.backWhile(lparenth - 1, JavaParserUtil.WS_COMMENTS);
if (tokens.hasType(i, GT)) {
i = tokens.backWhile(tokens.backWithBraceMatching(i, LT, GT), JavaParserUtil.WS_COMMENTS);
@@ -8,6 +8,7 @@ import com.intellij.lang.impl.TokenSequence;
import com.intellij.lang.java.JavaLanguage;
import com.intellij.lang.java.JavaParserDefinition;
import com.intellij.lexer.Lexer;
import com.intellij.lexer.TokenizedText;
import com.intellij.openapi.project.Project;
import com.intellij.openapi.util.Condition;
import com.intellij.openapi.util.Key;
@@ -39,7 +40,7 @@ public class JavaParserUtil {
private static final Key<Boolean> DEEP_PARSE_BLOCKS_IN_STATEMENTS = Key.create("JavaParserUtil.ParserExtender");
@NotNull
public static TokenSequence obtainTokens(@NotNull PsiFile file) {
public static TokenizedText obtainTokens(@NotNull PsiFile file) {
return CachedValuesManager.getCachedValue(file, () ->
CachedValueProvider.Result.create(
TokenSequence.performLexing(file.getViewProvider().getContents(), JavaParserDefinition.createLexer(PsiUtil.getLanguageLevel(file))),
@@ -22,6 +22,7 @@ import com.intellij.lang.LighterLazyParseableNode;
import com.intellij.lang.impl.TokenSequence;
import com.intellij.lang.java.JavaParserDefinition;
import com.intellij.lexer.Lexer;
import com.intellij.lexer.TokenizedText;
import com.intellij.psi.JavaTokenType;
import com.intellij.psi.PsiFile;
import com.intellij.psi.PsiJavaFile;
@@ -69,7 +70,7 @@ public class JavaLightStubBuilder extends LightStubBuilder {
CodeBlockVisitor visitor = new CodeBlockVisitor();
if (TreeUtil.isCollapsedChameleon(node)) {
Lexer lexer = JavaParserDefinition.createLexer(PsiUtil.getLanguageLevel(node.getPsi()));
TokenSequence tokens = TokenSequence.performLexing(node.getChars(), lexer);
TokenizedText tokens = TokenSequence.performLexing(node.getChars(), lexer);
for (int i = 0; i < tokens.getTokenCount(); i++) {
visitor.visit(tokens.getTokenType(i));
}
@@ -0,0 +1,192 @@
// Copyright 2000-2020 JetBrains s.r.o. Use of this source code is governed by the Apache 2.0 license that can be found in the LICENSE file.
package com.intellij.lexer;
import com.intellij.openapi.util.Comparing;
import com.intellij.psi.tree.IElementType;
import com.intellij.psi.tree.TokenSet;
import com.intellij.util.ArrayUtil;
import org.jetbrains.annotations.ApiStatus;
import org.jetbrains.annotations.NotNull;
import org.jetbrains.annotations.Nullable;
/**
* This class represents the result of lexing: text and the tokens produced from it by some lexer.
* It allows clients to inspect all tokens at once and easily move back and forward to implement some simple lexer-based checks.
*/
@ApiStatus.Experimental
public interface TokenizedText {
/**
* @return the number of tokens inside
*/
int getTokenCount();
/**
* @return the full text that was split into the tokens represented here
*/
@NotNull
CharSequence getText();
/**
* @return the start offset of the token with the given index
*/
int getTokenStart(int index);
/**
* @return the end offset of the token with the given index
*/
int getTokenEnd(int index);
/**
* @return the type of the token with the given index, or null if the index is negative or exceeds token count
*/
IElementType getTokenType(int index);
/**
* @return the text of the token with the given index, or null if the index is negative or exceeds token count
*/
default CharSequence getTokenText(int index) {
if (index < 0 || index >= getTokenCount()) return null;
return getText().subSequence(getTokenStart(index), getTokenEnd(index));
}
/**
* @return whether {@link #getTokenType}(index) would return the given type
*/
default boolean hasType(int index, @NotNull IElementType type) {
return getTokenType(index) == type;
}
/**
* @return whether {@link #getTokenType}(index) would return any of the given types (null acceptable, indicating start or end of the text)
*/
default boolean hasType(int index, @Nullable IElementType @NotNull ... types) {
return ArrayUtil.contains(getTokenType(index), types);
}
/**
* @return whether {@link #getTokenType}(index) would return a type in the given set
*/
default boolean hasType(int index, @NotNull TokenSet types) {
return types.contains(getTokenType(index));
}
/**
* Moves back, potentially skipping tokens which represent a valid nesting sequence
* with the given types for opening and closing braces.
* @return an index {@code prev} of a token before {@code index} such that either:
* <ol>
* <li>{@code prev == index - 1}</li>
* <li>{@code hasType(prev + 1, opening) && hasType(index, closing)} and every opening brace between those indices has its closing one before {@code index}</li>
* </ol>
*/
default int backWithBraceMatching(int index, @NotNull IElementType opening, @NotNull IElementType closing) {
if (getTokenType(index) == closing) {
int nesting = 1;
while (nesting > 0 && index > 0) {
index--;
IElementType type = getTokenType(index);
if (type == closing) {
nesting++;
}
else if (type == opening) {
nesting--;
}
}
}
return index - 1;
}
/**
* Moves back from {@code index} while tokens belong to the given set
* @return the largest {@code prev <= index} whose token type doesn't belong to {@code toSkip}
*/
default int backWhile(int index, @NotNull TokenSet toSkip) {
while (hasType(index, toSkip)) {
index--;
}
return index;
}
/**
* Moves forward from {@code index} while tokens belong to the given set
* @return the smallest {@code next >= index} whose token type doesn't belong to {@code toSkip}
*/
default int forwardWhile(int index, @NotNull TokenSet toSkip) {
while (hasType(index, toSkip)) {
index++;
}
return index;
}
/**
* @return a view of this object as a {@link Lexer}.
* Note that the returned lexer isn't the same as the one that produced this tokenized text: it returns the same offsets and types,
* but states and positions might differ. The returned lexer may be used to avoid tokenizing the same text again in APIs where lexer is expected,
* but it will only accept the very same text from the very beginning; it can't be used on any other strings.
*/
default @NotNull Lexer asLexer() {
return new WrappingLexer(this);
}
/**
* A simple lexer over {@link TokenizedText}.
*/
class WrappingLexer extends LexerBase {
private final TokenizedText myTokens;
private int myIndex;
WrappingLexer(TokenizedText tokens) {
this.myTokens = tokens;
}
public TokenizedText getTokens() {
return myTokens;
}
@Override
public void start(@NotNull CharSequence buffer, int startOffset, int endOffset, int initialState) {
assert Comparing.equal(buffer, myTokens.getText());
assert startOffset == 0;
assert endOffset == buffer.length();
assert initialState == 0;
myIndex = 0;
}
@Override
public int getState() {
return myIndex;
}
@Override
public @Nullable IElementType getTokenType() {
return myTokens.getTokenType(myIndex);
}
@Override
public int getTokenStart() {
return myTokens.getTokenStart(myIndex);
}
@Override
public int getTokenEnd() {
return myTokens.getTokenEnd(myIndex);
}
@Override
public void advance() {
myIndex++;
}
@Override
public @NotNull CharSequence getBufferSequence() {
return myTokens.getText();
}
@Override
public int getBufferEnd() {
return myTokens.getText().length();
}
}
}
@@ -16,19 +16,18 @@
package com.intellij.lang.impl;
import com.intellij.lexer.Lexer;
import com.intellij.lexer.LexerBase;
import com.intellij.lexer.TokenizedText;
import com.intellij.openapi.diagnostic.Logger;
import com.intellij.openapi.progress.ProgressIndicatorProvider;
import com.intellij.openapi.util.Comparing;
import com.intellij.psi.tree.IElementType;
import com.intellij.psi.tree.TokenSet;
import com.intellij.util.ArrayUtil;
import org.jetbrains.annotations.ApiStatus;
import org.jetbrains.annotations.NotNull;
import org.jetbrains.annotations.Nullable;
@ApiStatus.Experimental
public class TokenSequence {
public class TokenSequence implements TokenizedText {
private static final Logger LOG = Logger.getInstance(TokenSequence.class);
private final CharSequence myText;
@@ -58,134 +57,41 @@ public class TokenSequence {
@NotNull
public static TokenSequence performLexing(@NotNull CharSequence text, @NotNull Lexer lexer) {
if (lexer instanceof WrappingLexer) {
TokenSequence existing = ((WrappingLexer)lexer).mySequence;
if (Comparing.equal(text, existing.myText)) {
TokenizedText existing = ((WrappingLexer)lexer).getTokens();
if (existing instanceof TokenSequence && Comparing.equal(text, ((TokenSequence)existing).myText)) {
// prevent clients like PsiBuilder from modifying shared token types
return new TokenSequence(existing.lexStarts, existing.lexTypes.clone(), existing.lexemeCount, text);
return new TokenSequence(((TokenSequence)existing).lexStarts,
((TokenSequence)existing).lexTypes.clone(),
((TokenSequence)existing).lexemeCount, text);
}
}
return new Builder(text, lexer).performLexing();
}
@Override
public int getTokenCount() {
return lexemeCount;
}
@Override
public @Nullable IElementType getTokenType(int index) {
if (index < 0 || index >= getTokenCount()) return null;
return lexTypes[index];
}
public boolean hasType(int index, @NotNull IElementType type) {
return getTokenType(index) == type;
}
public boolean hasType(int index, @Nullable IElementType @NotNull ... types) {
return ArrayUtil.contains(getTokenType(index), types);
}
public boolean hasType(int index, @NotNull TokenSet types) {
return types.contains(getTokenType(index));
}
public int backWithBraceMatching(int index, @NotNull IElementType opening, @NotNull IElementType closing) {
if (getTokenType(index) == closing) {
int nesting = 1;
while (nesting > 0 && index > 0) {
index--;
IElementType type = getTokenType(index);
if (type == closing) {
nesting++;
}
else if (type == opening) {
nesting--;
}
}
}
return index - 1;
}
public int backWhile(int index, @NotNull TokenSet toSkip) {
while (hasType(index, toSkip)) {
index--;
}
return index;
}
public int forwardWhile(int index, @NotNull TokenSet toSkip) {
while (hasType(index, toSkip)) {
index++;
}
return index;
}
@Override
public int getTokenStart(int index) {
return lexStarts[index];
}
@Override
public int getTokenEnd(int index) {
return lexStarts[index + 1];
}
public CharSequence getTokenText(int index) {
return myText.subSequence(getTokenStart(index), getTokenEnd(index));
}
public @NotNull Lexer asLexer() {
return new WrappingLexer(this);
}
private static class WrappingLexer extends LexerBase {
final TokenSequence mySequence;
private int myIndex;
WrappingLexer(TokenSequence sequence) {
this.mySequence = sequence;
}
@Override
public void start(@NotNull CharSequence buffer, int startOffset, int endOffset, int initialState) {
assert buffer.equals(mySequence.myText);
assert startOffset == 0;
assert endOffset == buffer.length();
assert initialState == 0;
myIndex = 0;
}
@Override
public int getState() {
return myIndex;
}
@Override
public @Nullable IElementType getTokenType() {
return mySequence.lexTypes[myIndex];
}
@Override
public int getTokenStart() {
return mySequence.lexStarts[myIndex];
}
@Override
public int getTokenEnd() {
return mySequence.lexStarts[myIndex + 1];
}
@Override
public void advance() {
myIndex++;
}
@Override
public @NotNull CharSequence getBufferSequence() {
return mySequence.myText;
}
@Override
public int getBufferEnd() {
return mySequence.myText.length();
}
@Override
public @NotNull CharSequence getText() {
return myText;
}
private static class Builder {