break words in complex identifiers when indexing (needed for CLJ-243)

This commit is contained in:
peter
2013-10-03 21:18:13 +02:00
parent 05a5e6eb4a
commit b9403aad5d
5 changed files with 65 additions and 28 deletions
@@ -247,6 +247,28 @@ public class FindManagerTest extends DaemonAnalyzerTestCase {
assertEquals(2, usages.size());
}
public void testDollars() throws Exception {
createFile(myModule, "A.java", "foo foo$ $foo");
createFile(myModule, "A.txt", "foo foo$ $foo");
FindModel findModel = new FindModel();
findModel.setWholeWordsOnly(true);
findModel.setFromCursor(false);
findModel.setGlobal(true);
findModel.setMultipleFiles(true);
findModel.setProjectScope(true);
findModel.setStringToFind("foo");
assertSize(2, findUsages(findModel));
findModel.setStringToFind("foo$");
assertSize(2, findUsages(findModel));
findModel.setStringToFind("$foo");
assertSize(2, findUsages(findModel));
}
public void testReplaceRegexp() throws Throwable {
FindManager findManager = FindManager.getInstance(myProject);
@@ -55,7 +55,7 @@ public class DefaultWordsScanner implements WordsScanner {
* @param identifierTokenSet the set of token types which represent identifiers.
* @param commentTokenSet the set of token types which represent comments.
* @param literalTokenSet the set of token types which represent literals.
* @param SkipCodeContextTokenSet the set of token types which should not be considered as code context.
* @param skipCodeContextTokenSet the set of token types which should not be considered as code context.
*/
public DefaultWordsScanner(final Lexer lexer, final TokenSet identifierTokenSet, final TokenSet commentTokenSet,
final TokenSet literalTokenSet, @NotNull TokenSet skipCodeContextTokenSet) {
@@ -68,19 +68,14 @@ public class DefaultWordsScanner implements WordsScanner {
public void processWords(CharSequence fileText, Processor<WordOccurrence> processor) {
myLexer.start(fileText);
WordOccurrence occurrence = null; // shared occurrence
WordOccurrence occurrence = new WordOccurrence(fileText, 0, 0, null); // shared occurrence
IElementType type;
while ((type = myLexer.getTokenType()) != null) {
if (myIdentifierTokenSet.contains(type)) {
if (occurrence == null) {
occurrence = new WordOccurrence(fileText, myLexer.getTokenStart(), myLexer.getTokenEnd(), WordOccurrence.Kind.CODE);
}
else {
occurrence.init(fileText, myLexer.getTokenStart(), myLexer.getTokenEnd(), WordOccurrence.Kind.CODE);
}
if (!processor.process(occurrence)) return;
}
//occurrence.init(fileText, myLexer.getTokenStart(), myLexer.getTokenEnd(), WordOccurrence.Kind.CODE);
//if (!processor.process(occurrence)) return;
if (!stripWords(processor, fileText, myLexer.getTokenStart(), myLexer.getTokenEnd(), WordOccurrence.Kind.CODE, occurrence, false)) return; }
else if (myCommentTokenSet.contains(type)) {
if (!stripWords(processor, fileText,myLexer.getTokenStart(),myLexer.getTokenEnd(), WordOccurrence.Kind.COMMENTS,occurrence, false)) return;
}
@@ -99,7 +94,7 @@ public class DefaultWordsScanner implements WordsScanner {
int from,
int to,
final WordOccurrence.Kind kind,
WordOccurrence occurence,
@NotNull WordOccurrence occurrence,
boolean mayHaveFileRefs
) {
// This code seems strange but it is more effective as Character.isJavaIdentifier_xxx_ is quite costly operation due to unicode
@@ -110,8 +105,7 @@ public class DefaultWordsScanner implements WordsScanner {
while (true) {
if (index == to) break ScanWordsLoop;
char c = tokenText.charAt(index);
if ((c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || (c >= '0' && c <= '9') ||
(c != '$' && Character.isJavaIdentifierStart(c))) {
if (isAsciiIdentifierPart(c) || Character.isJavaIdentifierStart(c)) {
break;
}
index++;
@@ -121,27 +115,26 @@ public class DefaultWordsScanner implements WordsScanner {
index++;
if (index == to) break;
char c = tokenText.charAt(index);
if ((c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || (c >= '0' && c <= '9')) continue;
if (c == '$' || !Character.isJavaIdentifierPart(c)) break;
if (isAsciiIdentifierPart(c)) continue;
if (!Character.isJavaIdentifierPart(c)) break;
}
int wordEnd = index;
if (occurence == null) {
occurence = new WordOccurrence(tokenText, wordStart, wordEnd, kind);
}
else {
occurence.init(tokenText, wordStart, wordEnd, kind);
}
occurrence.init(tokenText, wordStart, wordEnd, kind);
if (!processor.process(occurence)) return false;
if (!processor.process(occurrence)) return false;
if (mayHaveFileRefs) {
occurence.init(tokenText,wordStart, wordEnd, WordOccurrence.Kind.FOREIGN_LANGUAGE);
if (!processor.process(occurence)) return false;
occurrence.init(tokenText, wordStart, wordEnd, WordOccurrence.Kind.FOREIGN_LANGUAGE);
if (!processor.process(occurrence)) return false;
}
}
return true;
}
private static boolean isAsciiIdentifierPart(char c) {
return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || (c >= '0' && c <= '9') || c == '$';
}
public void setMayHaveFileRefsInLiterals(final boolean mayHaveFileRefsInLiterals) {
myMayHaveFileRefsInLiterals = mayHaveFileRefsInLiterals;
}
@@ -87,7 +87,7 @@ public class IdIndex extends FileBasedIndexExtension<IdIndexEntry, Integer> {
@Override
public int getVersion() {
return 9; // TODO: version should enumerate all word scanner versions and build version upon that set
return 10; // TODO: version should enumerate all word scanner versions and build version upon that set
}
@Override
@@ -17,6 +17,7 @@
package com.intellij.psi.impl.cache.impl.id;
import com.intellij.ide.highlighter.custom.CustomFileTypeLexer;
import com.intellij.ide.highlighter.custom.SyntaxTable;
import com.intellij.lang.Language;
import com.intellij.lang.cacheBuilder.*;
import com.intellij.lang.findUsages.FindUsagesProvider;
@@ -107,14 +108,14 @@ public class IdTableBuilding {
}
if (fileType instanceof CustomSyntaxTableFileType) {
return new WordsScannerFileTypeIdIndexerAdapter(createWordScanner((CustomSyntaxTableFileType)fileType));
return new WordsScannerFileTypeIdIndexerAdapter(createCustomFileTypeScanner(((CustomSyntaxTableFileType)fileType).getSyntaxTable()));
}
return null;
}
private static WordsScanner createWordScanner(final CustomSyntaxTableFileType customSyntaxTableFileType) {
return new DefaultWordsScanner(new CustomFileTypeLexer(customSyntaxTableFileType.getSyntaxTable(), true),
public static WordsScanner createCustomFileTypeScanner(SyntaxTable syntaxTable) {
return new DefaultWordsScanner(new CustomFileTypeLexer(syntaxTable, true),
TokenSet.create(CustomHighlighterTokenType.IDENTIFIER),
TokenSet.create(CustomHighlighterTokenType.LINE_COMMENT,
CustomHighlighterTokenType.MULTI_LINE_COMMENT),
@@ -15,11 +15,15 @@
* limitations under the License.
*/
package com.intellij.ide.highlighter.custom
import com.intellij.lang.cacheBuilder.WordOccurrence
import com.intellij.lexer.Lexer
import com.intellij.openapi.fileTypes.PlainTextSyntaxHighlighterFactory
import com.intellij.openapi.util.text.StringUtil
import com.intellij.psi.impl.cache.impl.id.IdTableBuilding
import com.intellij.testFramework.LexerTestCase
import com.intellij.testFramework.PlatformTestUtil
import com.intellij.util.Processor
import com.intellij.util.ThrowableRunnable
import junit.framework.TestCase
import org.jetbrains.annotations.NonNls
@@ -331,6 +335,23 @@ KEYWORD_2 ('foo{}')
'''
}
public void testWordsScanner() {
SyntaxTable table = new SyntaxTable()
table.addKeyword1("a*")
def scanner = IdTableBuilding.createCustomFileTypeScanner(table)
def words = []
String text = 'a* b-c d# e$ foo{}'
def expectedWords = ['a', 'b', 'c', 'd', 'e$', 'foo']
scanner.processWords(text, { WordOccurrence w ->
words.add(w.baseText.subSequence(w.start, w.end))
} as Processor)
assert words == expectedWords
// words searched by find usages should be the same as words produced by word scanner
assert StringUtil.getWordsIn(text) == expectedWords
}
public void "test quote block comment"() {
SyntaxTable table = new SyntaxTable()
table.startComment = '"'