diff --git a/platform/util/src/com/intellij/openapi/util/text/Pluralizer.java b/platform/util/src/com/intellij/openapi/util/text/Pluralizer.java new file mode 100644 index 000000000000..5474e2d78b8e --- /dev/null +++ b/platform/util/src/com/intellij/openapi/util/text/Pluralizer.java @@ -0,0 +1,407 @@ +/* + * The original license from http://github.com/blakeembrey/pluralize: + * + * The MIT License (MIT) + * + * Copyright (c) 2013 Blake Embrey (hello@blakeembrey.com) + * + * Permission is hereby granted, free of charge, to any person obtaining a copy + * of this software and associated documentation files (the "Software"), to deal + * in the Software without restriction, including without limitation the rights + * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + * copies of the Software, and to permit persons to whom the Software is + * furnished to do so, subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in + * all copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN + * THE SOFTWARE. + */ +package com.intellij.openapi.util.text; + +import com.intellij.openapi.util.Pair; +import com.intellij.util.Consumer; +import com.intellij.util.containers.ContainerUtil; +import com.intellij.util.containers.JBIterable; +import com.intellij.util.text.CaseInsensitiveStringHashingStrategy; + +import java.util.List; +import java.util.Map; +import java.util.Set; +import java.util.regex.Matcher; +import java.util.regex.Pattern; + +/** + * A java version of http://github.com/blakeembrey/pluralize + * Revision: 90d82f88428f057c4b7a1d46aa38fc7c44d2d869 + * + * It tries to preserve the original structure for future sync. + * + * Other sources of inspiration: + * http://www.csse.monash.edu.au/~damian/papers/HTML/Plurals.html + * (a java implementation: https://github.com/atteo/evo-inflector) + * + * @author gregsh + */ +class Pluralizer { + + static final Pluralizer PLURALIZER; + + private final Map irregularSingles = ContainerUtil.newTroveMap(CaseInsensitiveStringHashingStrategy.INSTANCE); + private final Map irregularPlurals = ContainerUtil.newTroveMap(CaseInsensitiveStringHashingStrategy.INSTANCE); + private final Set uncountables = ContainerUtil.newTroveSet(CaseInsensitiveStringHashingStrategy.INSTANCE); + private final List> pluralRules = ContainerUtil.newArrayList(); + private final List> singularRules = ContainerUtil.newArrayList(); + + /** + * Pass in a word token to produce a function that can replicate the case on + * another word. + */ + private static String restoreCase(String word, String result) { + if (word == null || result == null || word == result) return result; + char[] chars = result.toCharArray(); + boolean prevUp = false; + int len = Math.min(chars.length, word.length()); + for (int i = 0; i < len; i++) { + char wc = word.charAt(i); + if (chars[i] == wc && i != len - 1) continue; + char uc = Character.toUpperCase(chars[i]); + char lc = Character.toLowerCase(chars[i]); + if (wc == lc || (prevUp = wc == uc)) { + chars[i] = wc; + } + } + for (int i = len; i < chars.length; i++) { + chars[i] = prevUp ? Character.toUpperCase(chars[i]) : Character.toLowerCase(chars[i]); + } + return new String(chars); + } + + /** + * Sanitize a word by passing in the word and sanitization rules. + */ + private String sanitizeWord(String word, List> rules) { + if (StringUtil.isEmpty(word) || uncountables.contains(word)) return word; + + int len = rules.size(); + + while (--len > -1) { + Pair rule = rules.get(len); + Matcher matcher = rule.first.matcher(word); + if (matcher.find()) { + return matcher.replaceFirst(rule.second); + } + } + return word; + } + + /** + * Replace a word with the updated word. + */ + private String replaceWord(String word, Map replaceMap, Map keepMap, List> rules) { + if (StringUtil.isEmpty(word)) return word; + + // Get the correct token and case restoration functions. + // Check against the keep object map. + if (keepMap.containsKey(word)) return word; + + // Check against the replacement map for a direct word replacement. + if (replaceMap.containsKey(word)) { + return restoreCase(word, replaceMap.get(word)); + } + + // Run all the rules against the word. + return sanitizeWord(word, rules); + } + + /** + * Pluralize or singularize a word based on the passed in count. + */ + public String pluralize(String word, int count, boolean inclusive) { + String pluralized = count == 1 ? singular(word) : plural(word); + + return (inclusive ? count + " " : "") + pluralized; + } + + public String plural(String word) { + return restoreCase(word, replaceWord(word, irregularSingles, irregularPlurals, pluralRules)); + } + + public String singular(String word) { + return restoreCase(word, replaceWord(word, irregularPlurals, irregularSingles, singularRules)); + } + + private static Pattern sanitizeRule(String rule) { + return Pattern.compile(rule.startsWith("/") ? rule.substring(1) : "^" + rule + "$", Pattern.CASE_INSENSITIVE); + } + + protected void addPluralRule(String rule, String replacement) { + pluralRules.add(Pair.create(sanitizeRule(rule), replacement)); + } + + protected void addSingularRule(String rule, String replacement) { + singularRules.add(Pair.create(sanitizeRule(rule), replacement)); + } + + protected void addUncountableRule(String word) { + if (!word.startsWith("/")) { + uncountables.add(word); + } + else { + // Set singular and plural references for the word. + addPluralRule(word, "$0"); + addSingularRule(word, "$0"); + } + } + + protected void addIrregularRule(String single, String plural) { + irregularSingles.put(single, plural); + irregularPlurals.put(plural, single); + } + + static { + final Pluralizer pluralizer = new Pluralizer(); + + /* + * Irregular rules. + */ + JBIterable.of(new String[][]{ + // Pronouns. + {"I", "we"}, + {"me", "us"}, + {"he", "they"}, + {"she", "they"}, + {"them", "them"}, + {"myself", "ourselves"}, + {"yourself", "yourselves"}, + {"itself", "themselves"}, + {"herself", "themselves"}, + {"himself", "themselves"}, + {"themself", "themselves"}, + {"is", "are"}, + {"was", "were"}, + {"has", "have"}, + {"this", "these"}, + {"that", "those"}, + // Words ending in with a consonant and `o`. + {"echo", "echoes"}, + {"dingo", "dingoes"}, + {"volcano", "volcanoes"}, + {"tornado", "tornadoes"}, + {"torpedo", "torpedoes"}, + // Ends with `us`. + {"genus", "genera"}, + {"viscus", "viscera"}, + // Ends with `ma`. + {"stigma", "stigmata"}, + {"stoma", "stomata"}, + {"dogma", "dogmata"}, + {"lemma", "lemmata"}, + {"schema", "schemata"}, + {"anathema", "anathemata"}, + // Other irregular rules. + {"ox", "oxen"}, + {"axe", "axes"}, + {"die", "dice"}, + {"yes", "yeses"}, + {"foot", "feet"}, + {"eave", "eaves"}, + {"goose", "geese"}, + {"tooth", "teeth"}, + {"quiz", "quizzes"}, + {"human", "humans"}, + {"proof", "proofs"}, + {"carve", "carves"}, + {"valve", "valves"}, + {"looey", "looies"}, + {"thief", "thieves"}, + {"groove", "grooves"}, + {"pickaxe", "pickaxes"}, + {"whiskey", "whiskies"} + }).consumeEach(new Consumer() { + @Override + public void consume(String[] o) { + pluralizer.addIrregularRule(o[0], o[1]); + } + }); + + /* + * Pluralization rules. + */ + JBIterable.of(new String[][]{ + {"/s?$", "s"}, + {"/([^aeiou]ese)$", "$1"}, + {"/(ax|test)is$", "$1es"}, + {"/(alias|[^aou]us|tlas|gas|ris)$", "$1es"}, + {"/(e[mn]u)s?$", "$1s"}, + {"/([^l]ias|[aeiou]las|[emjzr]as|[iu]am)$", "$1"}, + {"/(alumn|syllab|octop|vir|radi|nucle|fung|cact|stimul|termin|bacill|foc|uter|loc|strat)(?:us|i)$", "$1i"}, + {"/(alumn|alg|vertebr)(?:a|ae)$", "$1ae"}, + {"/(seraph|cherub)(?:im)?$", "$1im"}, + {"/(her|at|gr)o$", "$1oes"}, + {"/(agend|addend|millenni|medi|dat|extrem|bacteri|desiderat|strat|candelabr|errat|ov|symposi|curricul|automat|quor)(?:a|um)$", "$1a"}, + {"/(apheli|hyperbat|periheli|asyndet|noumen|phenomen|criteri|organ|prolegomen|hedr|automat)(?:a|on)$", "$1a"}, + {"/sis$", "ses"}, + {"/(?:(kni|wi|li)fe|(ar|l|ea|eo|oa|hoo)f)$", "$1$2ves"}, + {"/([^aeiouy]|qu)y$", "$1ies"}, + {"/([^ch][ieo][ln])ey$", "$1ies"}, + {"/(x|ch|ss|sh|zz)$", "$1es"}, + {"/(matr|cod|mur|sil|vert|ind|append)(?:ix|ex)$", "$1ices"}, + {"/(m|l)(?:ice|ouse)$", "$1ice"}, + {"/(pe)(?:rson|ople)$", "$1ople"}, + {"/(child)(?:ren)?$", "$1ren"}, + {"/eaux$", "$0"}, + {"/m[ae]n$", "men"}, + {"thou", "you"} + }).consumeEach(new Consumer() { + @Override + public void consume(String[] o) { + pluralizer.addPluralRule(o[0], o[1]); + } + }); + + /* + * Singularization rules. + */ + JBIterable.of(new String[][]{ + {"/(.)s$", "$1"}, + {"/(ss)$", "$1"}, + {"/((a)naly|(b)a|(d)iagno|(p)arenthe|(p)rogno|(s)ynop|(t)he)(?:sis|ses)$", "$1sis"}, + {"/(^analy)(?:sis|ses)$", "$1sis"}, + {"/(wi|kni|(?:after|half|high|low|mid|non|night|[^\\w]|^)li)ves$", "$1fe"}, + {"/(ar|(?:wo|[ae])l|[eo][ao])ves$", "$1f"}, + {"/ies$", "y"}, + {"/\\b([pl]|zomb|(?:neck|cross)?t|coll|faer|food|gen|goon|group|lass|talk|goal|cut)ies$", "$1ie"}, + {"/\\b(mon|smil)ies$", "$1ey"}, + {"/(m|l)ice$", "$1ouse"}, + {"/(seraph|cherub)im$", "$1"}, + {"/(x|ch|ss|sh|zz|tto|go|cho|alias|[^aou]us|tlas|gas|(?:her|at|gr)o|ris)(?:es)?$", "$1"}, + {"/(e[mn]u)s?$", "$1"}, + {"/(cookie|movie|twelve)s$", "$1"}, + {"/(cris|test|diagnos)(?:is|es)$", "$1is"}, + {"/(alumn|syllab|octop|vir|radi|nucle|fung|cact|stimul|termin|bacill|foc|uter|loc|strat)(?:us|i)$", "$1us"}, + {"/(agend|addend|millenni|dat|extrem|bacteri|desiderat|strat|candelabr|errat|ov|symposi|curricul|quor)a$", "$1um"}, + {"/(apheli|hyperbat|periheli|asyndet|noumen|phenomen|criteri|organ|prolegomen|hedr|automat)a$", "$1on"}, + {"/(alumn|alg|vertebr)ae$", "$1a"}, + {"/(cod|mur|sil|vert|ind)ices$", "$1ex"}, + {"/(matr|append)ices$", "$1ix"}, + {"/(pe)(rson|ople)$", "$1rson"}, + {"/(child)ren$", "$1"}, + {"/(eau)x?$", "$1"}, + {"/men$", "man"} + }).consumeEach(new Consumer() { + @Override + public void consume(String[] o) { + pluralizer.addSingularRule(o[0], o[1]); + } + }); + /* + * Uncountable rules. + */ + JBIterable.of( + // Singular words with no plurals. + "advice", + "adulthood", + "agenda", + "aid", + "alcohol", + "ammo", + "athletics", + "bison", + "blood", + "bream", + "buffalo", + "butter", + "carp", + "cash", + "chassis", + "chess", + "clothing", + "commerce", + "cod", + "cooperation", + "corps", + "digestion", + "debris", + "diabetes", + "energy", + "equipment", + "elk", + "excretion", + "expertise", + "flounder", + "fun", + "gallows", + "garbage", + "graffiti", + "headquarters", + "health", + "herpes", + "highjinks", + "homework", + "housework", + "information", + "jeans", + "justice", + "kudos", + "labour", + "literature", + "machinery", + "mackerel", + "mail", + "media", + "mews", + "moose", + "music", + "news", + "pike", + "plankton", + "pliers", + "pollution", + "premises", + "rain", + "research", + "rice", + "salmon", + "scissors", + "series", + "sewage", + "shambles", + "shrimp", + "species", + "staff", + "swine", + "trout", + "traffic", + "transportation", + "tuna", + "wealth", + "welfare", + "whiting", + "wildebeest", + "wildlife", + "you", + // Regexes. + "/pox$", // "chickpox", "smallpox" + "/ois$", + "/deer$", // "deer", "reindeer" + "/fish$", // "fish", "blowfish", "angelfish" + "/sheep$", + "/measles$", + "/[^aeiou]ese$/i" // "chinese", "japanese" + ).consumeEach(new Consumer() { + @Override + public void consume(String o) { + pluralizer.addUncountableRule(o); + } + }); + + PLURALIZER = pluralizer; + } +} diff --git a/platform/util/src/com/intellij/openapi/util/text/StringUtil.java b/platform/util/src/com/intellij/openapi/util/text/StringUtil.java index e525c567970a..4e9b313020fb 100644 --- a/platform/util/src/com/intellij/openapi/util/text/StringUtil.java +++ b/platform/util/src/com/intellij/openapi/util/text/StringUtil.java @@ -829,49 +829,10 @@ public class StringUtil extends StringUtilRt { if (escaped) buffer.append('\\'); } - @SuppressWarnings("HardCodedStringLiteral") @NotNull @Contract(pure = true) - public static String pluralize(@NotNull String suggestion) { - if (suggestion.endsWith("Child") || suggestion.endsWith("child")) { - return suggestion + "ren"; - } - - if (suggestion.equals("this")) { - return "these"; - } - if (suggestion.equals("This")) { - return "These"; - } - if (suggestion.equals("fix") || suggestion.equals("Fix")) { - return suggestion + "es"; - } - - if (endsWithIgnoreCase(suggestion, "es")) { - return suggestion; - } - - int len = suggestion.length(); - if (endsWithIgnoreCase(suggestion, "ex") || endsWithIgnoreCase(suggestion, "ix")) { - return suggestion.substring(0, len - 2) + "ices"; - } - if (endsWithIgnoreCase(suggestion, "um")) { - return suggestion.substring(0, len - 2) + "a"; - } - if (endsWithIgnoreCase(suggestion, "an")) { - return suggestion.substring(0, len - 2) + "en"; - } - - if (endsWithIgnoreCase(suggestion, "s") || endsWithIgnoreCase(suggestion, "x") || - endsWithIgnoreCase(suggestion, "ch") || endsWithIgnoreCase(suggestion, "sh")) { - return suggestion + "es"; - } - - if (endsWithIgnoreCase(suggestion, "y") && len > 1 && !isVowel(toLowerCase(suggestion.charAt(len - 2)))) { - return suggestion.substring(0, len - 1) + "ies"; - } - - return suggestion + "s"; + public static String pluralize(@NotNull String word) { + return Pluralizer.PLURALIZER.plural(word); } @NotNull @@ -1687,55 +1648,10 @@ public class StringUtil extends StringUtilRt { * @param name english word in plural form * @return name in singular form or null if failed to find one. */ - @SuppressWarnings("HardCodedStringLiteral") @Nullable @Contract(pure = true) - public static String unpluralize(@NotNull final String name) { - if (name.endsWith("sses") || name.endsWith("shes") || name.endsWith("ches") || name.endsWith("xes")) { //? - return name.substring(0, name.length() - 2); - } - - if (name.endsWith("ses")) { - return name.substring(0, name.length() - 1); - } - - if (name.endsWith("ies")) { - if (name.endsWith("cookies") || name.endsWith("Cookies")) { - return name.substring(0, name.length() - "ookies".length()) + "ookie"; - } - - return name.substring(0, name.length() - 3) + "y"; - } - - if (name.endsWith("leaves") || name.endsWith("Leaves")) { - return name.substring(0, name.length() - "eaves".length()) + "eaf"; - } - - String result = stripEnding(name, "s"); - if (result != null) { - return result; - } - - if (name.endsWith("children")) { - return name.substring(0, name.length() - "children".length()) + "child"; - } - - if (name.endsWith("Children") && name.length() > "Children".length()) { - return name.substring(0, name.length() - "Children".length()) + "Child"; - } - - - return null; - } - - @Nullable - @Contract(pure = true) - private static String stripEnding(@NotNull String name, @NotNull String ending) { - if (name.endsWith(ending)) { - if (name.equals(ending)) return name; // do not return empty string - return name.substring(0, name.length() - 1); - } - return null; + public static String unpluralize(@NotNull String word) { + return Pluralizer.PLURALIZER.singular(word); } @Contract(pure = true) diff --git a/platform/util/testSrc/com/intellij/util/text/StringUtilTest.java b/platform/util/testSrc/com/intellij/util/text/StringUtilTest.java index 5c10e1184e36..e1594bc26518 100644 --- a/platform/util/testSrc/com/intellij/util/text/StringUtilTest.java +++ b/platform/util/testSrc/com/intellij/util/text/StringUtilTest.java @@ -100,6 +100,12 @@ public class StringUtilTest { public void testUnPluralize() { assertEquals("s", StringUtil.unpluralize("s")); assertEquals("z", StringUtil.unpluralize("zs")); + assertEquals("Index", StringUtil.unpluralize("Indices")); + assertEquals("fix", StringUtil.unpluralize("fixes")); + assertEquals("man", StringUtil.unpluralize("men")); + assertEquals("leaf", StringUtil.unpluralize("leaves")); + assertEquals("cookie", StringUtil.unpluralize("cookies")); + assertEquals("search", StringUtil.unpluralize("searches")); } @Test @@ -112,6 +118,13 @@ public class StringUtilTest { assertEquals("men", StringUtil.pluralize("man")); assertEquals("media", StringUtil.pluralize("medium")); assertEquals("stashes", StringUtil.pluralize("stash")); + assertEquals("children", StringUtil.pluralize("child")); + assertEquals("leaves", StringUtil.pluralize("leaf")); + assertEquals("These", StringUtil.pluralize("This")); + assertEquals("cookies", StringUtil.pluralize("cookie")); + assertEquals("VaLuES", StringUtil.pluralize("VaLuE")); + assertEquals("PLANS", StringUtil.pluralize("PLAN")); + assertEquals("stackTraceLineExes", StringUtil.pluralize("stackTraceLineEx")); } @Test