@@ -208,18 +208,44 @@ private static List<String> getMainTokenStrs(String[] tokens) {
208208 return mainTokenStrs ;
209209 }
210210
211+ /**
212+ * Filters a list of tokens down to the "main" tokens: the nonempty ones
213+ * which are at least 4 characters long or start with an uppercase letter.
214+ *
215+ * @param tokens The tokens to filter
216+ * @return A new list containing the main tokens, in their original order
217+ */
211218 public static List <String > getMainStrs (List <String > tokens ) {
212219 List <String > mainTokenStrs = new ArrayList <>(tokens .size ());
213220 mainTokenStrs .addAll (tokens .stream ().filter (text -> !text .isEmpty () && (text .length () >= 4 || Character .isUpperCase (text .charAt (0 )))).collect (Collectors .toList ()));
214221 return mainTokenStrs ;
215222 }
216223
224+ /**
225+ * Returns true if the given string is an acronym of the given tokens.
226+ *
227+ * @param str The candidate acronym
228+ * @param tokens The tokens of the candidate expansion
229+ * @return true if {@code str} is an acronym of {@code tokens}
230+ * @see AcronymMatcher#isAcronymImpl(String, List)
231+ */
217232 public static boolean isAcronym (String str , String [] tokens ) {
218233 return isAcronymImpl (str , Arrays .asList (tokens ));
219234 }
220235
221236 // Public static utility methods
222237
238+ /**
239+ * Returns true if the given string is an acronym of the given tokens.
240+ * The characters '-', '.' and '_' are first removed from {@code str}; if its length
241+ * then differs from the number of tokens, stopwords are removed from the tokens.
242+ * It is an acronym if each character matches (case-insensitively) the first
243+ * character of the corresponding token. Empty tokens match any character.
244+ *
245+ * @param str The candidate acronym
246+ * @param tokens The tokens of the candidate expansion
247+ * @return true if {@code str} is an acronym of {@code tokens}
248+ */
223249 public static boolean isAcronymImpl (String str , List <String > tokens ) {
224250 // Remove some words from the candidate acronym
225251 str = discardPattern .matcher (str ).replaceAll ("" );
@@ -242,6 +268,16 @@ public static boolean isAcronymImpl(String str, List<String> tokens) {
242268 }
243269 }
244270
271+ /**
272+ * Returns true if the given string is an acronym of the given tokens.
273+ * Tokens which are {@link CoreMap}s contribute their text annotation;
274+ * any other object contributes its {@code toString()}.
275+ *
276+ * @param str The candidate acronym
277+ * @param tokens The tokens of the candidate expansion
278+ * @return true if {@code str} is an acronym of {@code tokens}
279+ * @see AcronymMatcher#isAcronymImpl(String, List)
280+ */
245281 public static boolean isAcronym (String str , List <?> tokens ) {
246282 List <String > strs = new ArrayList <>(tokens .size ());
247283 for (Object tok : tokens ) {
@@ -259,6 +295,8 @@ public static boolean isAcronym(String str, List<?> tokens) {
259295 /**
260296 * Returns true if either chunk1 or chunk2 is acronym of the other.
261297 *
298+ * @param chunk1 The first chunk, with text and tokens annotations
299+ * @param chunk2 The second chunk, with text and tokens annotations
262300 * @return true if either chunk1 or chunk2 is acronym of the other
263301 */
264302 public static boolean isAcronym (CoreMap chunk1 , CoreMap chunk2 ) {
@@ -276,7 +314,15 @@ public static boolean isAcronym(CoreMap chunk1, CoreMap chunk2) {
276314 return isAcro ;
277315 }
278316
279- /** @see AcronymMatcher#isAcronym(edu.stanford.nlp.util.CoreMap, edu.stanford.nlp.util.CoreMap) */
317+ /**
318+ * Returns true if either chunk1 or chunk2 is acronym of the other.
319+ * The text of each chunk is its tokens joined with spaces.
320+ *
321+ * @param chunk1 The tokens of the first chunk
322+ * @param chunk2 The tokens of the second chunk
323+ * @return true if either chunk1 or chunk2 is acronym of the other
324+ * @see AcronymMatcher#isAcronym(edu.stanford.nlp.util.CoreMap, edu.stanford.nlp.util.CoreMap)
325+ */
280326 public static boolean isAcronym (String [] chunk1 , String [] chunk2 ) {
281327 String text1 = StringUtils .join (chunk1 );
282328 String text2 = StringUtils .join (chunk2 );
@@ -292,6 +338,15 @@ public static boolean isAcronym(String[] chunk1, String[] chunk2) {
292338 return isAcro ;
293339 }
294340
341+ /**
342+ * Returns true if either chunk1 or chunk2 is a "fancy" acronym of the other,
343+ * as determined by {@link #isFancyAcronymImpl(String, List)}.
344+ * The text of each chunk is its tokens joined with spaces.
345+ *
346+ * @param chunk1 The tokens of the first chunk
347+ * @param chunk2 The tokens of the second chunk
348+ * @return true if either chunk is a fancy acronym of the other
349+ */
295350 public static boolean isFancyAcronym (String [] chunk1 , String [] chunk2 ) {
296351 String text1 = StringUtils .join (chunk1 );
297352 String text2 = StringUtils .join (chunk2 );
@@ -301,6 +356,14 @@ public static boolean isFancyAcronym(String[] chunk1, String[] chunk2) {
301356 return isFancyAcronymImpl (text1 , tokenStrs2 ) || isFancyAcronymImpl (text2 , tokenStrs1 );
302357 }
303358
359+ /**
360+ * Returns true if the characters of {@code str} (after removing '-', '.' and '_')
361+ * occur in order, case-sensitively, in the space-joined text of the tokens.
362+ *
363+ * @param str The candidate acronym
364+ * @param tokens The tokens of the candidate expansion
365+ * @return true if {@code str} is a fancy acronym of {@code tokens}
366+ */
304367 public static boolean isFancyAcronymImpl (String str , List <String > tokens ) {
305368 str = discardPattern .matcher (str ).replaceAll ("" );
306369 String text = StringUtils .join (tokens );
0 commit comments