mirror of
https://gitflic.ru/project/openide/openide.git
synced 2026-09-27 10:03:11 +07:00
use isjavaidentifierpart + enable trigramindex for tests because test appeared
This commit is contained in:
@@ -57,8 +57,6 @@ import com.intellij.util.Processor;
|
||||
import com.intellij.util.containers.ContainerUtil;
|
||||
import com.intellij.util.indexing.FileBasedIndex;
|
||||
import com.intellij.util.indexing.FileBasedIndexImpl;
|
||||
import gnu.trove.TIntHashSet;
|
||||
import gnu.trove.TIntIterator;
|
||||
import org.jetbrains.annotations.NotNull;
|
||||
import org.jetbrains.annotations.Nullable;
|
||||
|
||||
@@ -363,7 +361,14 @@ class FindInProjectTask {
|
||||
String text = myFindModel.getStringToFind();
|
||||
if (StringUtil.isEmptyOrSpaces(text)) return false;
|
||||
|
||||
if (TrigramIndex.ENABLED) return !TrigramBuilder.buildTrigram(text).isEmpty();
|
||||
if (TrigramIndex.ENABLED) {
|
||||
return !TrigramBuilder.processTrigrams(text, new TrigramBuilder.TrigramProcessor() {
|
||||
@Override
|
||||
public boolean execute(int value) {
|
||||
return false;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// $ is used to separate words when indexing plain-text files but not when indexing
|
||||
// Java identifiers, so we can't consistently break a string containing $ characters into words
|
||||
@@ -394,12 +399,14 @@ class FindInProjectTask {
|
||||
final Set<PsiFile> resultFiles = new LinkedHashSet<PsiFile>();
|
||||
|
||||
if (TrigramIndex.ENABLED) {
|
||||
Set<Integer> keys = ContainerUtil.newTroveSet();
|
||||
TIntHashSet trigrams = TrigramBuilder.buildTrigram(stringToFind);
|
||||
TIntIterator it = trigrams.iterator();
|
||||
while (it.hasNext()) {
|
||||
keys.add(it.next());
|
||||
}
|
||||
final Set<Integer> keys = ContainerUtil.newTroveSet();
|
||||
TrigramBuilder.processTrigrams(stringToFind, new TrigramBuilder.TrigramProcessor() {
|
||||
@Override
|
||||
public boolean execute(int value) {
|
||||
keys.add(value);
|
||||
return true;
|
||||
}
|
||||
});
|
||||
|
||||
if (!keys.isEmpty()) {
|
||||
List<VirtualFile> hits = new ArrayList<VirtualFile>();
|
||||
|
||||
@@ -19,7 +19,6 @@
|
||||
*/
|
||||
package com.intellij.find.ngrams;
|
||||
|
||||
import com.intellij.openapi.application.ApplicationManager;
|
||||
import com.intellij.openapi.util.ThreadLocalCachedIntArray;
|
||||
import com.intellij.openapi.util.text.TrigramBuilder;
|
||||
import com.intellij.openapi.vfs.VirtualFile;
|
||||
@@ -31,8 +30,6 @@ import com.intellij.util.io.DataInputOutputUtil;
|
||||
import com.intellij.util.io.EnumeratorIntegerDescriptor;
|
||||
import com.intellij.util.io.KeyDescriptor;
|
||||
import gnu.trove.THashMap;
|
||||
import gnu.trove.TIntHashSet;
|
||||
import gnu.trove.TIntProcedure;
|
||||
import org.jetbrains.annotations.NotNull;
|
||||
|
||||
import java.io.DataInput;
|
||||
@@ -44,8 +41,7 @@ import java.util.Collection;
|
||||
import java.util.Map;
|
||||
|
||||
public class TrigramIndex extends ScalarIndexExtension<Integer> implements CustomInputsIndexFileBasedIndexExtension<Integer> {
|
||||
public static final boolean ENABLED = SystemProperties.getBooleanProperty("idea.internal.trigramindex.enabled",
|
||||
!ApplicationManager.getApplication().isUnitTestMode());
|
||||
public static final boolean ENABLED = SystemProperties.getBooleanProperty("idea.internal.trigramindex.enabled", true);
|
||||
|
||||
public static final ID<Integer,Void> INDEX_ID = ID.create("Trigram.Index");
|
||||
|
||||
@@ -75,16 +71,10 @@ public class TrigramIndex extends ScalarIndexExtension<Integer> implements Custo
|
||||
@Override
|
||||
@NotNull
|
||||
public Map<Integer, Void> map(@NotNull FileContent inputData) {
|
||||
TIntHashSet built = TrigramBuilder.buildTrigram(inputData.getContentAsText());
|
||||
final Map<Integer, Void> result = new THashMap<Integer, Void>(built.size());
|
||||
built.forEach(new TIntProcedure() {
|
||||
@Override
|
||||
public boolean execute(int value) {
|
||||
result.put(value, null);
|
||||
return true;
|
||||
}
|
||||
});
|
||||
return result;
|
||||
MyTrigramProcessor trigramProcessor = new MyTrigramProcessor();
|
||||
TrigramBuilder.processTrigrams(inputData.getContentAsText(), trigramProcessor);
|
||||
|
||||
return trigramProcessor.map;
|
||||
}
|
||||
};
|
||||
}
|
||||
@@ -160,4 +150,19 @@ public class TrigramIndex extends ScalarIndexExtension<Integer> implements Custo
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
private static class MyTrigramProcessor extends TrigramBuilder.TrigramProcessor {
|
||||
Map<Integer, Void> map;
|
||||
@Override
|
||||
public boolean consumeTrigramsCount(int count) {
|
||||
map = new THashMap<Integer, Void>(count);
|
||||
return true;
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean execute(int value) {
|
||||
map.put(value, null);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+30
-127
@@ -15,141 +15,44 @@
|
||||
*/
|
||||
package com.intellij.openapi.util.text;
|
||||
|
||||
import com.intellij.openapi.util.io.FileUtil;
|
||||
import com.intellij.openapi.util.io.FileUtilRt;
|
||||
import gnu.trove.THashSet;
|
||||
import gnu.trove.TIntHashSet;
|
||||
import gnu.trove.TIntObjectHashMap;
|
||||
import gnu.trove.TIntObjectProcedure;
|
||||
import com.intellij.openapi.util.Ref;
|
||||
import gnu.trove.TIntArrayList;
|
||||
import junit.framework.TestCase;
|
||||
|
||||
import java.io.BufferedReader;
|
||||
import java.io.File;
|
||||
import java.io.FileReader;
|
||||
import java.io.IOException;
|
||||
import java.util.*;
|
||||
public class TrigramBuilderTest extends TestCase {
|
||||
public void testBuilder() {
|
||||
final Ref<Integer> trigramCountRef = new Ref<Integer>();
|
||||
final TIntArrayList list = new TIntArrayList();
|
||||
|
||||
public class TrigramBuilderTest {
|
||||
public static void main(String[] args) throws IOException {
|
||||
File root = new File(args[0]);
|
||||
TrigramBuilder.processTrigrams("String$CharData", new TrigramBuilder.TrigramProcessor() {
|
||||
@Override
|
||||
public boolean execute(int value) {
|
||||
list.add(value);
|
||||
return true;
|
||||
}
|
||||
|
||||
Stats stats = new Stats();
|
||||
walk(root, stats);
|
||||
|
||||
System.out.println("Scanned " + stats.files + " files, total of " + stats.lines + " lines in " + (stats.time / 1000000) + " ms.");
|
||||
System.out.println("Size:" + stats.bytes);
|
||||
System.out.println("Total trigrams: " + stats.allTrigrams.size());
|
||||
System.out.println("Max per file: " + stats.maxtrigrams);
|
||||
|
||||
System.out.println("Sample query 1: " + lookup(stats.filesMap, "trigram"));
|
||||
System.out.println("Sample query 2: " + lookup(stats.filesMap, "some text that most probably doesn't exist"));
|
||||
System.out.println("Sample query 3: " + lookup(stats.filesMap, "ProfilingUtil.captureCPUSnapshot();"));
|
||||
|
||||
System.out.println("Stop words:");
|
||||
|
||||
listWithBarier(stats, stats.files * 2 / 4);
|
||||
listWithBarier(stats, stats.files * 3 / 4);
|
||||
listWithBarier(stats, stats.files * 4 / 5);
|
||||
}
|
||||
|
||||
private static void listWithBarier(Stats stats, final int barrier) {
|
||||
final int[] stopCount = {0};
|
||||
stats.filesMap.forEachEntry(new TIntObjectProcedure<List<File>>() {
|
||||
public boolean execute(int a, List<File> b) {
|
||||
if (b.size() > barrier) {
|
||||
System.out.println(a);
|
||||
stopCount[0]++;
|
||||
}
|
||||
@Override
|
||||
public boolean consumeTrigramsCount(int count) {
|
||||
trigramCountRef.set(count);
|
||||
return true;
|
||||
}
|
||||
});
|
||||
|
||||
System.out.println("Total of " + stopCount[0]);
|
||||
list.sort();
|
||||
Integer trigramCount = trigramCountRef.get();
|
||||
assertNotNull(trigramCount);
|
||||
|
||||
int expectedTrigramCount = 13;
|
||||
assertEquals(expectedTrigramCount, (int)trigramCount);
|
||||
assertEquals(expectedTrigramCount, list.size());
|
||||
|
||||
int[] expected = {buildTrigram("$Ch"), buildTrigram("arD"), buildTrigram("ata"), 6514785, 6578548, 6759523, 6840690, 6909543, 7235364, 7496801, 7498094, 7566450, 7631465, };
|
||||
for(int i = 0; i < expectedTrigramCount; ++i) assertEquals(expected[i], list.getQuick(i));
|
||||
}
|
||||
|
||||
private static void walk(File root, Stats stats) throws IOException {
|
||||
String name = root.getName();
|
||||
if (root.isDirectory()) {
|
||||
if (name.startsWith(".") || name.equals("out")) return;
|
||||
System.out.println("Lexing in " + root.getPath());
|
||||
for (File file : root.listFiles()) {
|
||||
walk(file, stats);
|
||||
}
|
||||
}
|
||||
else {
|
||||
String ext = FileUtilRt.getExtension(name);
|
||||
if (!allowedExtension.contains(ext)) return;
|
||||
if (root.length() > 100 * 1024) return;
|
||||
|
||||
stats.extensions.add(ext);
|
||||
lex(root, stats);
|
||||
}
|
||||
private static int buildTrigram(String s) {
|
||||
int tc1 = StringUtil.toLowerCase(s.charAt(0));
|
||||
int tc2 = (tc1 << 8) + StringUtil.toLowerCase(s.charAt(1));
|
||||
return (tc2 << 8) + StringUtil.toLowerCase(s.charAt(2));
|
||||
}
|
||||
|
||||
private static void lex(File root, Stats stats) throws IOException {
|
||||
stats.files++;
|
||||
BufferedReader reader = new BufferedReader(new FileReader(root));
|
||||
String s;
|
||||
StringBuilder buf = new StringBuilder();
|
||||
while ((s = reader.readLine()) != null) {
|
||||
stats.lines++;
|
||||
buf.append(s).append("\n");
|
||||
}
|
||||
|
||||
stats.bytes += buf.length();
|
||||
|
||||
long start = System.nanoTime();
|
||||
TIntHashSet localTrigrams = lexText(buf);
|
||||
stats.time += System.nanoTime() - start;
|
||||
stats.maxtrigrams = Math.max(stats.maxtrigrams, localTrigrams.size());
|
||||
int[] graphs = localTrigrams.toArray();
|
||||
stats.allTrigrams.addAll(graphs);
|
||||
for (int graph : graphs) {
|
||||
List<File> list = stats.filesMap.get(graph);
|
||||
if (list == null) {
|
||||
list = new ArrayList<File>();
|
||||
stats.filesMap.put(graph, list);
|
||||
}
|
||||
list.add(root);
|
||||
}
|
||||
}
|
||||
|
||||
private static final Set<String> allowedExtension = new THashSet<String>(
|
||||
Arrays.asList("iml", "xml", "java", "html", "bat", "policy", "properties", "sh", "dtd", "ipr", "txt", "plist", "form", "xsl", "css",
|
||||
"jsp", "jspx", "xhtml", "tld", "htm", "tag", "jspf", "js", "ft", "xsd", "xls", "rb", "php", "ftl", "c", "y", "erb", "rjs",
|
||||
"rhtml", "sql", "cfml", "groovy", "text", "gsp", "h", "cc", "cpp", "wsdl"), FileUtil.PATH_HASHING_STRATEGY);
|
||||
|
||||
private static Collection<File> lookup(TIntObjectHashMap<List<File>> trigramsDatabase, String query) {
|
||||
final Set<File> result = new HashSet<File>();
|
||||
int[] graphs = TrigramBuilder.buildTrigram(query).toArray();
|
||||
boolean first = true;
|
||||
|
||||
for (int graph : graphs) {
|
||||
if (first) {
|
||||
result.addAll(trigramsDatabase.get(graph));
|
||||
first = false;
|
||||
}
|
||||
else {
|
||||
result.retainAll(trigramsDatabase.get(graph));
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
|
||||
private static TIntHashSet lexText(StringBuilder buf) {
|
||||
return TrigramBuilder.buildTrigram(buf);
|
||||
}
|
||||
|
||||
private static class Stats {
|
||||
public int files;
|
||||
public int lines;
|
||||
public int maxtrigrams;
|
||||
public long time;
|
||||
public long bytes;
|
||||
public final TIntHashSet allTrigrams = new TIntHashSet();
|
||||
public final TIntObjectHashMap<List<File>> filesMap = new TIntObjectHashMap<List<File>>();
|
||||
public final Set<String> extensions = new HashSet<String>();
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -20,15 +20,14 @@
|
||||
package com.intellij.openapi.util.text;
|
||||
|
||||
import com.intellij.util.text.CharArrayUtil;
|
||||
import gnu.trove.TIntHashSet;
|
||||
import gnu.trove.TIntProcedure;
|
||||
|
||||
public class TrigramBuilder {
|
||||
private TrigramBuilder() {
|
||||
}
|
||||
|
||||
public static TIntHashSet buildTrigram(CharSequence text) {
|
||||
TIntHashSet result = new TIntHashSet();
|
||||
|
||||
public static boolean processTrigrams(CharSequence text, TrigramProcessor consumer) {
|
||||
final AddonlyIntSet set = new AddonlyIntSet();
|
||||
int index = 0;
|
||||
final char[] fileTextArray = CharArrayUtil.fromSequenceWithoutCopying(text);
|
||||
|
||||
@@ -43,32 +42,123 @@ public class TrigramBuilder {
|
||||
}
|
||||
index++;
|
||||
}
|
||||
int index1 = index;
|
||||
int identifierStart = index;
|
||||
while (true) {
|
||||
index++;
|
||||
if (index == text.length()) break;
|
||||
final char c = fileTextArray != null ? fileTextArray[index]:text.charAt(index);
|
||||
if ((c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || (c >= '0' && c <= '9')) continue;
|
||||
if (!Character.isUnicodeIdentifierPart(c)) break;
|
||||
if (!Character.isJavaIdentifierPart(c)) break;
|
||||
}
|
||||
if (index - index1 > 100) continue; // Strange limit but we should have some!
|
||||
|
||||
int tc1 = 0;
|
||||
int tc2 = 0;
|
||||
int tc3;
|
||||
for (int i = index1, iters = 0; i < index; ++i, ++iters) {
|
||||
for (int i = identifierStart, iters = 0; i < index; ++i, ++iters) {
|
||||
char c = StringUtil.toLowerCase(fileTextArray != null ? fileTextArray[i]:text.charAt(i));
|
||||
tc3 = (tc2 << 8) + c;
|
||||
tc2 = (tc1 << 8) + c;
|
||||
tc1 = c;
|
||||
|
||||
if (iters >= 2) {
|
||||
result.add(tc3);
|
||||
set.add(tc3);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
return consumer.consumeTrigramsCount(set.size()) && set.forEach(consumer);
|
||||
}
|
||||
|
||||
public static abstract class TrigramProcessor implements TIntProcedure {
|
||||
public boolean consumeTrigramsCount(int count) { return true; }
|
||||
}
|
||||
}
|
||||
|
||||
class AddonlyIntSet {
|
||||
//private static final int MAGIC = 0x9E3779B9;
|
||||
private int size;
|
||||
private int[] data;
|
||||
private int shift;
|
||||
private int mask;
|
||||
private boolean hasZeroKey;
|
||||
|
||||
public AddonlyIntSet() {
|
||||
this(21);
|
||||
}
|
||||
|
||||
public AddonlyIntSet(int expectedSize) {
|
||||
int powerOfTwo = Integer.highestOneBit((3 * expectedSize) / 2) << 1;
|
||||
shift = Integer.numberOfLeadingZeros(powerOfTwo) + 1;
|
||||
mask = powerOfTwo - 1;
|
||||
data = new int[powerOfTwo];
|
||||
}
|
||||
|
||||
public int size() {
|
||||
return size;
|
||||
}
|
||||
|
||||
private int hash(int h, int[] a) {
|
||||
h ^= (h >>> 20) ^ (h >>> 12);
|
||||
return (h ^ (h >>> 7) ^ (h >>> 4)) & mask;
|
||||
//int idx = (id * MAGIC) >>> shift;
|
||||
//if (idx >= a.length) {
|
||||
// idx %= a.length;
|
||||
//}
|
||||
//return idx;
|
||||
}
|
||||
|
||||
public void add(int key) {
|
||||
if (key == 0) {
|
||||
if (!hasZeroKey) ++size;
|
||||
hasZeroKey = true;
|
||||
return;
|
||||
}
|
||||
if (size >= (2 * data.length) / 3) rehash();
|
||||
if (doPut(data, key)) size++;
|
||||
}
|
||||
|
||||
private boolean doPut(int[] a, int o) {
|
||||
int index = hash(o, a);
|
||||
int obj;
|
||||
while ((obj = a[index]) != 0) {
|
||||
if (obj == o) break;
|
||||
if (index == 0) index = a.length;
|
||||
index--;
|
||||
}
|
||||
a[index] = o;
|
||||
return obj == 0;
|
||||
}
|
||||
|
||||
private void rehash() {
|
||||
--shift;
|
||||
int[] b = new int[data.length << 1];
|
||||
mask = b.length - 1;
|
||||
for (int i = data.length; --i >= 0;) {
|
||||
int ns = data[i];
|
||||
if (ns != 0) doPut(b, ns);
|
||||
}
|
||||
data = b;
|
||||
}
|
||||
|
||||
public boolean contains(int key) {
|
||||
if (key == 0) return hasZeroKey;
|
||||
int index = hash(key, data);
|
||||
int v;
|
||||
while ((v = data[index]) != 0) {
|
||||
if (v == key) return true;
|
||||
if (index == 0) index = data.length;
|
||||
index--;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
public boolean forEach(TIntProcedure consumer) {
|
||||
if (hasZeroKey && !consumer.execute(0)) return false;
|
||||
for(int o:data) {
|
||||
if (o == 0) continue;
|
||||
if(!consumer.execute(o)) return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user