001/*
002 * Licensed to the Apache Software Foundation (ASF) under one or more
003 * contributor license agreements.  See the NOTICE file distributed with
004 * this work for additional information regarding copyright ownership.
005 * The ASF licenses this file to You under the Apache License, Version 2.0
006 * (the "License"); you may not use this file except in compliance with
007 * the License.  You may obtain a copy of the License at
008 *
009 *      https://www.apache.org/licenses/LICENSE-2.0
010 *
011 * Unless required by applicable law or agreed to in writing, software
012 * distributed under the License is distributed on an "AS IS" BASIS,
013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
014 * See the License for the specific language governing permissions and
015 * limitations under the License.
016 */
017package org.apache.commons.text;
018
019import java.util.HashSet;
020import java.util.Locale;
021import java.util.Set;
022import java.util.function.IntUnaryOperator;
023import java.util.function.Predicate;
024import java.util.regex.Matcher;
025import java.util.regex.Pattern;
026
027import org.apache.commons.lang3.ArrayUtils;
028import org.apache.commons.lang3.StringUtils;
029import org.apache.commons.lang3.Strings;
030import org.apache.commons.lang3.Validate;
031
032/**
033 * Operations on Strings that contain words.
034 *
035 * <p>
036 * This class tries to handle {@code null} input gracefully. An exception will not be thrown for a
037 * {@code null} input. Each method documents its behavior in more detail.
038 * </p>
039 *
040 * @since 1.1
041 */
042public class WordUtils {
043
044    /**
045     * Abbreviates the words nicely.
046     * <p>
047     * This method searches for the first space after the lower limit and abbreviates the String there. It will also append any String passed as a parameter to
048     * the end of the String. The upper limit can be specified to forcibly abbreviate a String.
049     * </p>
050     *
051     * <pre>
052     * WordUtils.abbreviate("Now is the time for all good men", 0, 40, null));     = "Now"
053     * WordUtils.abbreviate("Now is the time for all good men", 10, 40, null));    = "Now is the"
054     * WordUtils.abbreviate("Now is the time for all good men", 20, 40, null));    = "Now is the time for all"
055     * WordUtils.abbreviate("Now is the time for all good men", 0, 40, ""));       = "Now"
056     * WordUtils.abbreviate("Now is the time for all good men", 10, 40, ""));      = "Now is the"
057     * WordUtils.abbreviate("Now is the time for all good men", 20, 40, ""));      = "Now is the time for all"
058     * WordUtils.abbreviate("Now is the time for all good men", 0, 40, " ..."));   = "Now ..."
059     * WordUtils.abbreviate("Now is the time for all good men", 10, 40, " ..."));  = "Now is the ..."
060     * WordUtils.abbreviate("Now is the time for all good men", 20, 40, " ..."));  = "Now is the time for all ..."
061     * WordUtils.abbreviate("Now is the time for all good men", 0, -1, ""));       = "Now"
062     * WordUtils.abbreviate("Now is the time for all good men", 10, -1, ""));      = "Now is the"
063     * WordUtils.abbreviate("Now is the time for all good men", 20, -1, ""));      = "Now is the time for all"
064     * WordUtils.abbreviate("Now is the time for all good men", 50, -1, ""));      = "Now is the time for all good men"
065     * WordUtils.abbreviate("Now is the time for all good men", 1000, -1, ""));    = "Now is the time for all good men"
066     * WordUtils.abbreviate("Now is the time for all good men", 9, -10, null));    = Throws {@link IllegalArgumentException}
067     * WordUtils.abbreviate("Now is the time for all good men", 10, 5, null));     = Throws {@link IllegalArgumentException}
068     * </pre>
069     *
070     * @param str         The string to be abbreviated. If null is passed, null is returned. If the empty String is passed, the empty string is returned.
071     * @param lower       The lower limit; negative value is treated as zero.
072     * @param upper       The upper limit; specify -1 if no limit is desired. The upper limit cannot be lower than the lower limit.
073     * @param appendToEnd The String to be appended to the end of the abbreviated string. This is appended ONLY if the string was indeed abbreviated. The append
074     *                    does not count towards the lower or upper limits.
075     * @return The abbreviated String.
076     */
077    public static String abbreviate(final String str, int lower, int upper, final String appendToEnd) {
078        Validate.isTrue(upper >= -1, "upper value cannot be less than -1");
079        Validate.isTrue(upper >= lower || upper == -1, "upper value is less than lower value");
080        if (StringUtils.isEmpty(str)) {
081            return str;
082        }
083        // if the lower value is greater than the length of the string,
084        // set to the length of the string
085        if (lower > str.length()) {
086            lower = str.length();
087        }
088        // if the upper value is -1 (i.e. no limit) or is greater
089        // than the length of the string, set to the length of the string
090        if (upper == -1 || upper > str.length()) {
091            upper = str.length();
092        }
093        final StringBuilder result = new StringBuilder();
094        final int index = Strings.CS.indexOf(str, " ", lower);
095        if (index == -1) {
096            result.append(str, 0, upper);
097            // only if abbreviation has occurred do we append the appendToEnd value
098            if (upper != str.length()) {
099                result.append(StringUtils.defaultString(appendToEnd));
100            }
101        } else {
102            result.append(str, 0, Math.min(index, upper));
103            result.append(StringUtils.defaultString(appendToEnd));
104        }
105        return result.toString();
106    }
107
108    /**
109     * Applies a function to the first character of each word in a String.
110     * <p>
111     * This is used by {@link #capitalize(String, char...)} and {@link #uncapitalize(String, char...)}. The {@code transform} function is applied to the first
112     * code point of each word; all other code points are passed through unchanged.
113     * </p>
114     *
115     * @param str        The String to transform, may be null.
116     * @param delimiters The set of characters to determine word boundaries, null means whitespace.
117     * @param transform  The casing function to apply to the first code point of each word (e.g., {@code Character::toTitleCase} or
118     *                   {@code Character::toLowerCase}).
119     * @return The transformed String, or {@code null}/{@code ""} if the input is null/empty.
120     */
121    private static String applyWordCaseTransform(final String str, final char[] delimiters, final IntUnaryOperator transform) {
122        if (StringUtils.isEmpty(str)) {
123            return str;
124        }
125        final Predicate<Integer> isDelimiter = generateIsDelimiterFunction(delimiters);
126        final int strLen = str.length();
127        final int[] newCodePoints = new int[strLen];
128        int outOffset = 0;
129        boolean transformNext = true;
130        for (int index = 0; index < strLen;) {
131            final int codePoint = str.codePointAt(index);
132            if (isDelimiter.test(codePoint)) {
133                transformNext = true;
134                newCodePoints[outOffset++] = codePoint;
135                index += Character.charCount(codePoint);
136            } else if (transformNext) {
137                final int transformed = transform.applyAsInt(codePoint);
138                newCodePoints[outOffset++] = transformed;
139                index += Character.charCount(transformed);
140                transformNext = false;
141            } else {
142                newCodePoints[outOffset++] = codePoint;
143                index += Character.charCount(codePoint);
144            }
145        }
146        return new String(newCodePoints, 0, outOffset);
147    }
148
149    /**
150     * Capitalizes all the whitespace separated words in a String. Only the first character of each word is changed. To convert the rest of each word to
151     * lowercase at the same time, use {@link #capitalizeFully(String)}.
152     * <p>
153     * Whitespace is defined by {@link Character#isWhitespace(char)}. A {@code null} input String returns {@code null}. Capitalization uses the Unicode title
154     * case, normally equivalent to upper case.
155     * </p>
156     *
157     * <pre>
158     * WordUtils.capitalize(null)        = null
159     * WordUtils.capitalize("")          = ""
160     * WordUtils.capitalize("i am FINE") = "I Am FINE"
161     * </pre>
162     *
163     * @param str The String to capitalize, may be null.
164     * @return A new Capitalized String, or {@code null} if null String input.
165     * @see #uncapitalize(String)
166     * @see #capitalizeFully(String)
167     */
168    public static String capitalize(final String str) {
169        return capitalize(str, null);
170    }
171
172    /**
173     * Capitalizes all the delimiter separated words in a String. Only the first character of each word is changed. To convert the rest of each word to
174     * lowercase at the same time, use {@link #capitalizeFully(String, char[])}.
175     * <p>
176     * The delimiters represent a set of characters understood to separate words. The first string character and the first non-delimiter character after a
177     * delimiter will be capitalized.
178     * </p>
179     * <p>
180     * A {@code null} input String returns {@code null}. Capitalization uses the Unicode title case, normally equivalent to upper case.
181     * </p>
182     *
183     * <pre>
184     * WordUtils.capitalize(null, *)            = null
185     * WordUtils.capitalize("", *)              = ""
186     * WordUtils.capitalize(*, new char[0])     = *
187     * WordUtils.capitalize("i am fine", null)  = "I Am Fine"
188     * WordUtils.capitalize("i aM.fine", {'.'}) = "I aM.Fine"
189     * WordUtils.capitalize("i am fine", new char[]{}) = "I am fine"
190     * </pre>
191     *
192     * @param str        The String to capitalize, may be null.
193     * @param delimiters The Set of characters to determine capitalization, null means whitespace.
194     * @return A new Capitalized String, or {@code null} if null String input.
195     * @see #uncapitalize(String)
196     * @see #capitalizeFully(String)
197     */
198    public static String capitalize(final String str, final char... delimiters) {
199        return applyWordCaseTransform(str, delimiters, Character::toTitleCase);
200    }
201
202    /**
203     * Converts all the whitespace separated words in a String into capitalized words,
204     * that is each word is made up of a titlecase character and then a series of
205     * lowercase characters.
206     *
207     * <p>Whitespace is defined by {@link Character#isWhitespace(char)}.
208     * A {@code null} input String returns {@code null}.
209     * Capitalization uses the Unicode title case, normally equivalent to
210     * upper case.</p>
211     *
212     * <pre>
213     * WordUtils.capitalizeFully(null)        = null
214     * WordUtils.capitalizeFully("")          = ""
215     * WordUtils.capitalizeFully("i am FINE") = "I Am Fine"
216     * </pre>
217     *
218     * @param str  The String to capitalize, may be null.
219     * @return A new capitalized String, or {@code null} if null String input.
220     */
221    public static String capitalizeFully(final String str) {
222        return capitalizeFully(str, null);
223    }
224
225    /**
226     * Converts all the delimiter separated words in a String into capitalized words,
227     * that is each word is made up of a titlecase character and then a series of
228     * lowercase characters.
229     *
230     * <p>The delimiters represent a set of characters understood to separate words.
231     * The first string character and the first non-delimiter character after a
232     * delimiter will be capitalized.</p>
233     *
234     * <p>A {@code null} input String returns {@code null}.
235     * Capitalization uses the Unicode title case, normally equivalent to
236     * upper case.</p>
237     *
238     * <pre>
239     * WordUtils.capitalizeFully(null, *)            = null
240     * WordUtils.capitalizeFully("", *)              = ""
241     * WordUtils.capitalizeFully(*, null)            = *
242     * WordUtils.capitalizeFully(*, new char[0])     = *
243     * WordUtils.capitalizeFully("i aM.fine", {'.'}) = "I am.Fine"
244     * </pre>
245     *
246     * @param str  The String to capitalize, may be null.
247     * @param delimiters  The Set of characters to determine capitalization, null means whitespace.
248     * @return A new capitalized String, or {@code null} if null String input.
249     */
250    public static String capitalizeFully(final String str, final char... delimiters) {
251        return StringUtils.isEmpty(str) ? str : capitalize(str.toLowerCase(Locale.ROOT), delimiters);
252    }
253
254    /**
255     * Checks if the String contains all words in the given array.
256     *
257     * <p>
258     * A {@code null} String will return {@code false}. A {@code null}, zero
259     * length search array or if one element of array is null will return {@code false}.
260     * </p>
261     *
262     * <pre>
263     * WordUtils.containsAllWords(null, *)            = false
264     * WordUtils.containsAllWords("", *)              = false
265     * WordUtils.containsAllWords(*, null)            = false
266     * WordUtils.containsAllWords(*, [])              = false
267     * WordUtils.containsAllWords("abcd", "ab", "cd") = false
268     * WordUtils.containsAllWords("abc def", "def", "abc") = true
269     * </pre>
270     *
271     * @param word The CharSequence to check, may be null.
272     * @param words The array of String words to search for, may be null.
273     * @return {@code true} if all search words are found, {@code false} otherwise.
274     */
275    public static boolean containsAllWords(final CharSequence word, final CharSequence... words) {
276        if (StringUtils.isEmpty(word) || ArrayUtils.isEmpty(words)) {
277            return false;
278        }
279        for (final CharSequence w : words) {
280            if (StringUtils.isBlank(w)) {
281                return false;
282            }
283            final Pattern p = Pattern.compile(".*\\b" + Pattern.quote(w.toString()) + "\\b.*", Pattern.DOTALL);
284            if (!p.matcher(word).matches()) {
285                return false;
286            }
287        }
288        return true;
289    }
290
291    /**
292     * Given the array of delimiters supplied; returns a function determining whether a character code point is a delimiter.
293     * The function provides O(1) lookup time.
294     * Whitespace is defined by {@link Character#isWhitespace(char)} and is used as the defaultvalue if delimiters is null.
295     *
296     * @param delimiters The set of characters to determine delimiters, null means whitespace.
297     * @return A Predicate<Integer> taking a code point value as an argument and returning true if a delimiter.
298     */
299    private static Predicate<Integer> generateIsDelimiterFunction(final char[] delimiters) {
300        final Predicate<Integer> isDelimiter;
301        if (delimiters == null || delimiters.length == 0) {
302            isDelimiter = delimiters == null ? Character::isWhitespace : c -> false;
303        } else {
304            final Set<Integer> delimiterSet = new HashSet<>();
305            for (int index = 0; index < delimiters.length; index++) {
306                delimiterSet.add(Character.codePointAt(delimiters, index));
307            }
308            isDelimiter = delimiterSet::contains;
309        }
310        return isDelimiter;
311    }
312
313    /**
314     * Extracts the initial characters from each word in the String.
315     * <p>
316     * All first characters after whitespace are returned as a new string. Their case is not changed.
317     * </p>
318     * <p>
319     * Whitespace is defined by {@link Character#isWhitespace(char)}. A {@code null} input String returns {@code null}.
320     * </p>
321     *
322     * <pre>
323     * WordUtils.initials(null)             = null
324     * WordUtils.initials("")               = ""
325     * WordUtils.initials("Ben John Lee")   = "BJL"
326     * WordUtils.initials("Ben J.Lee")      = "BJ"
327     * </pre>
328     *
329     * @param str The String to get initials from, may be null.
330     * @return A new String of initial letters, or {@code null} if null String input.
331     * @see #initials(String,char[])
332     */
333    public static String initials(final String str) {
334        return initials(str, null);
335    }
336
337    /**
338     * Extracts the initial characters from each word in the String.
339     * <p>
340     * All first characters after the defined delimiters are returned as a new string. Their case is not changed.
341     * </p>
342     * <p>
343     * If the delimiters array is null, then Whitespace is used. Whitespace is defined by {@link Character#isWhitespace(char)}. A {@code null} input String
344     * returns {@code null}. An empty delimiter array returns an empty String.
345     * </p>
346     *
347     * <pre>
348     * WordUtils.initials(null, *)                = null
349     * WordUtils.initials("", *)                  = ""
350     * WordUtils.initials("Ben John Lee", null)   = "BJL"
351     * WordUtils.initials("Ben J.Lee", null)      = "BJ"
352     * WordUtils.initials("Ben J.Lee", [' ','.']) = "BJL"
353     * WordUtils.initials(*, new char[0])         = ""
354     * </pre>
355     *
356     * @param str        The String to get initials from, may be null.
357     * @param delimiters The Set of characters to determine words, null means whitespace.
358     * @return String of initial characters, or {@code null} if null String input.
359     * @see #initials(String)
360     */
361    public static String initials(final String str, final char... delimiters) {
362        if (StringUtils.isEmpty(str)) {
363            return str;
364        }
365        if (delimiters != null && delimiters.length == 0) {
366            return StringUtils.EMPTY;
367        }
368        final Predicate<Integer> isDelimiter = generateIsDelimiterFunction(delimiters);
369        final int strLen = str.length();
370        final int[] newCodePoints = new int[strLen / 2 + 1];
371        int count = 0;
372        boolean lastWasGap = true;
373        for (int i = 0; i < strLen;) {
374            final int codePoint = str.codePointAt(i);
375            if (isDelimiter.test(codePoint)) {
376                lastWasGap = true;
377            } else if (lastWasGap) {
378                newCodePoints[count++] = codePoint;
379                lastWasGap = false;
380            }
381            i += Character.charCount(codePoint);
382        }
383        return new String(newCodePoints, 0, count);
384    }
385
386    /**
387     * Is the character a delimiter.
388     *
389     * @param ch         The character to check.
390     * @param delimiters The delimiters.
391     * @return true if it is a delimiter.
392     * @deprecated as of 1.2 and will be removed in 2.0.
393     */
394    @Deprecated
395    public static boolean isDelimiter(final char ch, final char[] delimiters) {
396        if (delimiters == null) {
397            return Character.isWhitespace(ch);
398        }
399        for (final char delimiter : delimiters) {
400            if (ch == delimiter) {
401                return true;
402            }
403        }
404        return false;
405    }
406
407    /**
408     * Is the codePoint a delimiter.
409     *
410     * @param codePoint  The codePint to check.
411     * @param delimiters The delimiters.
412     * @return true if it is a delimiter.
413     * @deprecated as of 1.2 and will be removed in 2.0.
414     */
415    @Deprecated
416    public static boolean isDelimiter(final int codePoint, final char[] delimiters) {
417        if (delimiters == null) {
418            return Character.isWhitespace(codePoint);
419        }
420        for (int index = 0; index < delimiters.length; index++) {
421            final int delimiterCodePoint = Character.codePointAt(delimiters, index);
422            if (delimiterCodePoint == codePoint) {
423                return true;
424            }
425        }
426        return false;
427    }
428
429    /**
430     * Swaps the case of a String using a word based algorithm.
431     * <ul>
432     * <li>Upper case character converts to Lower case</li>
433     * <li>Title case character converts to Lower case</li>
434     * <li>Lower case character after Whitespace or at start converts to Title case</li>
435     * <li>Other Lower case character converts to Upper case</li>
436     * </ul>
437     * <p>
438     * Whitespace is defined by {@link Character#isWhitespace(char)}. A {@code null} input String returns {@code null}.
439     * </p>
440     *
441     * <pre>
442     * StringUtils.swapCase(null)                 = null
443     * StringUtils.swapCase("")                   = ""
444     * StringUtils.swapCase("The dog has a BONE") = "tHE DOG HAS A bone"
445     * </pre>
446     *
447     * @param str The String to swap case, may be null.
448     * @return The changed String, or {@code null} if null String input.
449     */
450    public static String swapCase(final String str) {
451        if (StringUtils.isEmpty(str)) {
452            return str;
453        }
454        final int strLen = str.length();
455        final int[] newCodePoints = new int[strLen];
456        int outOffset = 0;
457        boolean whitespace = true;
458        for (int index = 0; index < strLen;) {
459            final int oldCodepoint = str.codePointAt(index);
460            final int newCodePoint;
461            if (Character.isUpperCase(oldCodepoint) || Character.isTitleCase(oldCodepoint)) {
462                newCodePoint = Character.toLowerCase(oldCodepoint);
463                whitespace = false;
464            } else if (Character.isLowerCase(oldCodepoint)) {
465                if (whitespace) {
466                    newCodePoint = Character.toTitleCase(oldCodepoint);
467                    whitespace = false;
468                } else {
469                    newCodePoint = Character.toUpperCase(oldCodepoint);
470                }
471            } else {
472                whitespace = Character.isWhitespace(oldCodepoint);
473                newCodePoint = oldCodepoint;
474            }
475            newCodePoints[outOffset++] = newCodePoint;
476            index += Character.charCount(newCodePoint);
477        }
478        return new String(newCodePoints, 0, outOffset);
479    }
480
481    /**
482     * Uncapitalizes all the whitespace separated words in a String. Only the first character of each word is changed.
483     * <p>
484     * Whitespace is defined by {@link Character#isWhitespace(char)}. A {@code null} input String returns {@code null}.
485     * </p>
486     *
487     * <pre>
488     * WordUtils.uncapitalize(null)        = null
489     * WordUtils.uncapitalize("")          = ""
490     * WordUtils.uncapitalize("I Am FINE") = "i am fINE"
491     * </pre>
492     *
493     * @param str The String to uncapitalize, may be null.
494     * @return A new uncapitalized String, or {@code null} if null String input.
495     * @see #capitalize(String)
496     */
497    public static String uncapitalize(final String str) {
498        return uncapitalize(str, null);
499    }
500
501    /**
502     * Uncapitalizes all the whitespace separated words in a String. Only the first character of each word is changed.
503     * <p>
504     * The delimiters represent a set of characters understood to separate words. The first string character and the first non-delimiter character after a
505     * delimiter will be uncapitalized.
506     * </p>
507     * <p>
508     * Whitespace is defined by {@link Character#isWhitespace(char)}. A {@code null} input String returns {@code null}.
509     * </p>
510     *
511     * <pre>
512     * WordUtils.uncapitalize(null, *)            = null
513     * WordUtils.uncapitalize("", *)              = ""
514     * WordUtils.uncapitalize(*, null)            = *
515     * WordUtils.uncapitalize(*, new char[0])     = *
516     * WordUtils.uncapitalize("I AM.FINE", {'.'}) = "i AM.fINE"
517     * WordUtils.uncapitalize("I am fine", new char[]{}) = "i am fine"
518     * </pre>
519     *
520     * @param str        The String to uncapitalize, may be null.
521     * @param delimiters set of characters to determine uncapitalization, null means whitespace.
522     * @return uncapitalized String, or {@code null} if null String input.
523     * @see #capitalize(String)
524     */
525    public static String uncapitalize(final String str, final char... delimiters) {
526        return applyWordCaseTransform(str, delimiters, Character::toLowerCase);
527    }
528
529    /**
530     * Wraps a single line of text, identifying words by {@code ' '}.
531     *
532     * <p>New lines will be separated by the system property line separator.
533     * Very long words, such as URLs will <em>not</em> be wrapped.</p>
534     *
535     * <p>Leading spaces on a new line are stripped.
536     * Trailing spaces are not stripped.</p>
537     *
538     * <table border="1">
539     *  <caption>Examples</caption>
540     *  <tr>
541     *   <th>input</th>
542     *   <th>wrapLength</th>
543     *   <th>result</th>
544     *  </tr>
545     *  <tr>
546     *   <td>null</td>
547     *   <td>*</td>
548     *   <td>null</td>
549     *  </tr>
550     *  <tr>
551     *   <td>""</td>
552     *   <td>*</td>
553     *   <td>""</td>
554     *  </tr>
555     *  <tr>
556     *   <td>"Here is one line of text that is going to be wrapped after 20 columns."</td>
557     *   <td>20</td>
558     *   <td>"Here is one line of\ntext that is going\nto be wrapped after\n20 columns."</td>
559     *  </tr>
560     *  <tr>
561     *   <td>"Click here to jump to the commons website - https://commons.apache.org"</td>
562     *   <td>20</td>
563     *   <td>"Click here to jump\nto the commons\nwebsite -\nhttps://commons.apache.org"</td>
564     *  </tr>
565     *  <tr>
566     *   <td>"Click here, https://commons.apache.org, to jump to the commons website"</td>
567     *   <td>20</td>
568     *   <td>"Click here,\nhttps://commons.apache.org,\nto jump to the\ncommons website"</td>
569     *  </tr>
570     * </table>
571     *
572     * (assuming that '\n' is the systems line separator)
573     *
574     * @param str  The String to be word wrapped, may be null.
575     * @param wrapLength  The column to wrap the words at, less than 1 is treated as 1.
576     * @return A line with newlines inserted, {@code null} if null input.
577     */
578    public static String wrap(final String str, final int wrapLength) {
579        return wrap(str, wrapLength, null, false);
580    }
581
582    /**
583     * Wraps a single line of text, identifying words by {@code ' '}.
584     *
585     * <p>Leading spaces on a new line are stripped.
586     * Trailing spaces are not stripped.</p>
587     *
588     * <table border="1">
589     *  <caption>Examples</caption>
590     *  <tr>
591     *   <th>input</th>
592     *   <th>wrapLength</th>
593     *   <th>newLineString</th>
594     *   <th>wrapLongWords</th>
595     *   <th>result</th>
596     *  </tr>
597     *  <tr>
598     *   <td>null</td>
599     *   <td>*</td>
600     *   <td>*</td>
601     *   <td>true/false</td>
602     *   <td>null</td>
603     *  </tr>
604     *  <tr>
605     *   <td>""</td>
606     *   <td>*</td>
607     *   <td>*</td>
608     *   <td>true/false</td>
609     *   <td>""</td>
610     *  </tr>
611     *  <tr>
612     *   <td>"Here is one line of text that is going to be wrapped after 20 columns."</td>
613     *   <td>20</td>
614     *   <td>"\n"</td>
615     *   <td>true/false</td>
616     *   <td>"Here is one line of\ntext that is going\nto be wrapped after\n20 columns."</td>
617     *  </tr>
618     *  <tr>
619     *   <td>"Here is one line of text that is going to be wrapped after 20 columns."</td>
620     *   <td>20</td>
621     *   <td>"&lt;br /&gt;"</td>
622     *   <td>true/false</td>
623     *   <td>"Here is one line of&lt;br /&gt;text that is going&lt;
624     *   br /&gt;to be wrapped after&lt;br /&gt;20 columns."</td>
625     *  </tr>
626     *  <tr>
627     *   <td>"Here is one line of text that is going to be wrapped after 20 columns."</td>
628     *   <td>20</td>
629     *   <td>null</td>
630     *   <td>true/false</td>
631     *   <td>"Here is one line of" + systemNewLine + "text that is going"
632     *   + systemNewLine + "to be wrapped after" + systemNewLine + "20 columns."</td>
633     *  </tr>
634     *  <tr>
635     *   <td>"Click here to jump to the commons website - https://commons.apache.org"</td>
636     *   <td>20</td>
637     *   <td>"\n"</td>
638     *   <td>false</td>
639     *   <td>"Click here to jump\nto the commons\nwebsite -\nhttps://commons.apache.org"</td>
640     *  </tr>
641     *  <tr>
642     *   <td>"Click here to jump to the commons website - https://commons.apache.org"</td>
643     *   <td>20</td>
644     *   <td>"\n"</td>
645     *   <td>true</td>
646     *   <td>"Click here to jump\nto the commons\nwebsite -\nhttps://commons.apach\ne.org"</td>
647     *  </tr>
648     * </table>
649     *
650     * @param str  The String to be word wrapped, may be null.
651     * @param wrapLength  The column to wrap the words at, less than 1 is treated as 1.
652     * @param newLineStr  The string to insert for a new line, {@code null} uses the system property line separator.
653     * @param wrapLongWords  true if long words (such as URLs) should be wrapped.
654     * @return A line with newlines inserted, {@code null} if null input.
655     */
656    public static String wrap(final String str, final int wrapLength, final String newLineStr, final boolean wrapLongWords) {
657        return wrap(str, wrapLength, newLineStr, wrapLongWords, " ");
658    }
659
660    /**
661     * Wraps a single line of text, identifying words by {@code wrapOn}.
662     *
663     * <p>Leading spaces on a new line are stripped.
664     * Trailing spaces are not stripped.</p>
665     *
666     * <table border="1">
667     *  <caption>Examples</caption>
668     *  <tr>
669     *   <th>input</th>
670     *   <th>wrapLength</th>
671     *   <th>newLineString</th>
672     *   <th>wrapLongWords</th>
673     *   <th>wrapOn</th>
674     *   <th>result</th>
675     *  </tr>
676     *  <tr>
677     *   <td>null</td>
678     *   <td>*</td>
679     *   <td>*</td>
680     *   <td>true/false</td>
681     *   <td>*</td>
682     *   <td>null</td>
683     *  </tr>
684     *  <tr>
685     *   <td>""</td>
686     *   <td>*</td>
687     *   <td>*</td>
688     *   <td>true/false</td>
689     *   <td>*</td>
690     *   <td>""</td>
691     *  </tr>
692     *  <tr>
693     *   <td>"Here is one line of text that is going to be wrapped after 20 columns."</td>
694     *   <td>20</td>
695     *   <td>"\n"</td>
696     *   <td>true/false</td>
697     *   <td>" "</td>
698     *   <td>"Here is one line of\ntext that is going\nto be wrapped after\n20 columns."</td>
699     *  </tr>
700     *  <tr>
701     *   <td>"Here is one line of text that is going to be wrapped after 20 columns."</td>
702     *   <td>20</td>
703     *   <td>"&lt;br /&gt;"</td>
704     *   <td>true/false</td>
705     *   <td>" "</td>
706     *   <td>"Here is one line of&lt;br /&gt;text that is going&lt;br /&gt;
707     *   to be wrapped after&lt;br /&gt;20 columns."</td>
708     *  </tr>
709     *  <tr>
710     *   <td>"Here is one line of text that is going to be wrapped after 20 columns."</td>
711     *   <td>20</td>
712     *   <td>null</td>
713     *   <td>true/false</td>
714     *   <td>" "</td>
715     *   <td>"Here is one line of" + systemNewLine + "text that is going"
716     *   + systemNewLine + "to be wrapped after" + systemNewLine + "20 columns."</td>
717     *  </tr>
718     *  <tr>
719     *   <td>"Click here to jump to the commons website - https://commons.apache.org"</td>
720     *   <td>20</td>
721     *   <td>"\n"</td>
722     *   <td>false</td>
723     *   <td>" "</td>
724     *   <td>"Click here to jump\nto the commons\nwebsite -\nhttps://commons.apache.org"</td>
725     *  </tr>
726     *  <tr>
727     *   <td>"Click here to jump to the commons website - https://commons.apache.org"</td>
728     *   <td>20</td>
729     *   <td>"\n"</td>
730     *   <td>true</td>
731     *   <td>" "</td>
732     *   <td>"Click here to jump\nto the commons\nwebsite -\nhttps://commons.apach\ne.org"</td>
733     *  </tr>
734     *  <tr>
735     *   <td>"flammable/inflammable"</td>
736     *   <td>20</td>
737     *   <td>"\n"</td>
738     *   <td>true</td>
739     *   <td>"/"</td>
740     *   <td>"flammable\ninflammable"</td>
741     *  </tr>
742     * </table>
743     *
744     * @param str  The String to be word wrapped, may be null.
745     * @param wrapLength  The column to wrap the words at, less than 1 is treated as 1.
746     * @param newLineStr  The string to insert for a new line, {@code null} uses the system property line separator.
747     * @param wrapLongWords  true if long words (such as URLs) should be wrapped.
748     * @param wrapOn Regex expression to be used as a breakable characters, if blank string is provided a space character will be used.
749     * @return A line with newlines inserted, {@code null} if null input.
750     */
751    public static String wrap(final String str, int wrapLength, String newLineStr, final boolean wrapLongWords, String wrapOn) {
752        if (str == null) {
753            return null;
754        }
755        if (newLineStr == null) {
756            newLineStr = System.lineSeparator();
757        }
758        if (wrapLength < 1) {
759            wrapLength = 1;
760        }
761        if (StringUtils.isBlank(wrapOn)) {
762            wrapOn = " ";
763        }
764        final Pattern patternToWrapOn = Pattern.compile(wrapOn);
765        final int inputLineLength = str.length();
766        int offset = 0;
767        final StringBuilder wrappedLine = new StringBuilder(inputLineLength + 32);
768        int matcherSize = -1;
769        while (offset < inputLineLength) {
770            int spaceToWrapAt = -1;
771            Matcher matcher = patternToWrapOn
772                    .matcher(str.substring(offset, Math.min((int) Math.min(Integer.MAX_VALUE, offset + wrapLength + 1L), inputLineLength)));
773            if (matcher.find()) {
774                if (matcher.start() == 0) {
775                    matcherSize = matcher.end();
776                    if (matcherSize != 0) {
777                        offset += matcher.end();
778                        continue;
779                    }
780                    offset += 1;
781                }
782                spaceToWrapAt = matcher.start() + offset;
783            }
784            // only last line without leading spaces is left
785            if (inputLineLength - offset <= wrapLength) {
786                break;
787            }
788            while (matcher.find()) {
789                spaceToWrapAt = matcher.start() + offset;
790            }
791            if (spaceToWrapAt >= offset) {
792                // normal case
793                wrappedLine.append(str, offset, spaceToWrapAt);
794                wrappedLine.append(newLineStr);
795                offset = spaceToWrapAt + 1;
796            } else // really long word or URL
797            if (wrapLongWords) {
798                if (matcherSize == 0) {
799                    offset--;
800                }
801                // wrap really long word one line at a time, but keep a surrogate pair whole
802                int wrapAt = wrapLength + offset;
803                if (Character.isHighSurrogate(str.charAt(wrapAt - 1)) && Character.isLowSurrogate(str.charAt(wrapAt))) {
804                    wrapAt++;
805                }
806                wrappedLine.append(str, offset, wrapAt);
807                wrappedLine.append(newLineStr);
808                offset = wrapAt;
809                matcherSize = -1;
810            } else {
811                // do not wrap really long word, just extend beyond limit
812                matcher = patternToWrapOn.matcher(str.substring(offset + wrapLength));
813                if (matcher.find()) {
814                    matcherSize = matcher.end() - matcher.start();
815                    spaceToWrapAt = matcher.start() + offset + wrapLength;
816                }
817                if (spaceToWrapAt >= 0) {
818                    if (matcherSize == 0 && offset != 0) {
819                        offset--;
820                    }
821                    wrappedLine.append(str, offset, spaceToWrapAt);
822                    wrappedLine.append(newLineStr);
823                    offset = spaceToWrapAt + 1;
824                } else {
825                    if (matcherSize == 0 && offset != 0) {
826                        offset--;
827                    }
828                    wrappedLine.append(str, offset, str.length());
829                    offset = inputLineLength;
830                    matcherSize = -1;
831                }
832            }
833        }
834        if (matcherSize == 0 && offset < inputLineLength) {
835            offset--;
836        }
837        // Whatever is left in line is short enough to just pass through
838        wrappedLine.append(str, offset, str.length());
839        return wrappedLine.toString();
840    }
841
842    /**
843     * {@code WordUtils} instances should NOT be constructed in standard programming. Instead, the class should be used as
844     * {@code WordUtils.wrap("foo bar", 20);}.
845     * <p>
846     * This constructor is public to permit tools that require a JavaBean instance to operate.
847     * </p>
848     */
849    public WordUtils() {
850    }
851 }