001/*
002 * Licensed to the Apache Software Foundation (ASF) under one or more
003 * contributor license agreements.  See the NOTICE file distributed with
004 * this work for additional information regarding copyright ownership.
005 * The ASF licenses this file to You under the Apache License, Version 2.0
006 * (the "License"); you may not use this file except in compliance with
007 * the License.  You may obtain a copy of the License at
008 *
009 *      https://www.apache.org/licenses/LICENSE-2.0
010 *
011 * Unless required by applicable law or agreed to in writing, software
012 * distributed under the License is distributed on an "AS IS" BASIS,
013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
014 * See the License for the specific language governing permissions and
015 * limitations under the License.
016 */
017package org.apache.commons.lang3.text;
018
019import java.util.regex.Matcher;
020import java.util.regex.Pattern;
021
022import org.apache.commons.lang3.ArrayUtils;
023import org.apache.commons.lang3.StringUtils;
024
025/**
026 * Operations on Strings that contain words.
027 *
028 * <p>
029 * This class tries to handle {@code null} input gracefully.
030 * An exception will not be thrown for a {@code null} input.
031 * Each method documents its behavior in more detail.
032 * </p>
033 *
034 * @since 2.0
035 * @deprecated As of <a href="https://commons.apache.org/proper/commons-lang/changes-report.html#a3.6">3.6</a>, use Apache Commons Text
036 * <a href="https://commons.apache.org/proper/commons-text/javadocs/api-release/org/apache/commons/text/WordUtils.html">
037 * WordUtils</a>.
038 */
039@Deprecated
040public class WordUtils {
041
042    /**
043     * Capitalizes all the whitespace separated words in a String.
044     * Only the first character of each word is changed. To convert the
045     * rest of each word to lowercase at the same time,
046     * use {@link #capitalizeFully(String)}.
047     *
048     * <p>
049     * Whitespace is defined by {@link Character#isWhitespace(char)}.
050     * A {@code null} input String returns {@code null}.
051     * Capitalization uses the Unicode title case, normally equivalent to
052     * upper case.
053     * </p>
054     *
055     * <pre>
056     * WordUtils.capitalize(null)        = null
057     * WordUtils.capitalize("")          = ""
058     * WordUtils.capitalize("i am FINE") = "I Am FINE"
059     * </pre>
060     *
061     * @param str  The String to capitalize, may be null.
062     * @return capitalized String, {@code null} if null String input.
063     * @see #uncapitalize(String)
064     * @see #capitalizeFully(String)
065     */
066    public static String capitalize(final String str) {
067        return capitalize(str, null);
068    }
069
070    /**
071     * Capitalizes all the delimiter separated words in a String.
072     * Only the first character of each word is changed. To convert the
073     * rest of each word to lowercase at the same time,
074     * use {@link #capitalizeFully(String, char[])}.
075     *
076     * <p>
077     * The delimiters represent a set of characters understood to separate words.
078     * The first string character and the first non-delimiter character after a
079     * delimiter will be capitalized.
080     * </p>
081     *
082     * <p>
083     * A {@code null} input String returns {@code null}.
084     * Capitalization uses the Unicode title case, normally equivalent to
085     * upper case.
086     * </p>
087     *
088     * <pre>
089     * WordUtils.capitalize(null, *)            = null
090     * WordUtils.capitalize("", *)              = ""
091     * WordUtils.capitalize(*, new char[0])     = *
092     * WordUtils.capitalize("i am fine", null)  = "I Am Fine"
093     * WordUtils.capitalize("i aM.fine", {'.'}) = "I aM.Fine"
094     * </pre>
095     *
096     * @param str  The String to capitalize, may be null.
097     * @param delimiters  set of characters to determine capitalization, null means whitespace.
098     * @return capitalized String, {@code null} if null String input.
099     * @see #uncapitalize(String)
100     * @see #capitalizeFully(String)
101     * @since 2.1
102     */
103    public static String capitalize(final String str, final char... delimiters) {
104        final int delimLen = delimiters == null ? -1 : delimiters.length;
105        if (StringUtils.isEmpty(str) || delimLen == 0) {
106            return str;
107        }
108        final char[] buffer = str.toCharArray();
109        boolean capitalizeNext = true;
110        for (int i = 0; i < buffer.length;) {
111            final int codePoint = Character.codePointAt(buffer, i);
112            if (isDelimiter(codePoint, delimiters)) {
113                capitalizeNext = true;
114            } else if (capitalizeNext) {
115                Character.toChars(Character.toTitleCase(codePoint), buffer, i);
116                capitalizeNext = false;
117            }
118            i += Character.charCount(codePoint);
119        }
120        return new String(buffer);
121    }
122
123    /**
124     * Converts all the whitespace separated words in a String into capitalized words,
125     * that is each word is made up of a titlecase character and then a series of
126     * lowercase characters.
127     *
128     * <p>
129     * Whitespace is defined by {@link Character#isWhitespace(char)}.
130     * A {@code null} input String returns {@code null}.
131     * Capitalization uses the Unicode title case, normally equivalent to
132     * upper case.
133     * </p>
134     *
135     * <pre>
136     * WordUtils.capitalizeFully(null)        = null
137     * WordUtils.capitalizeFully("")          = ""
138     * WordUtils.capitalizeFully("i am FINE") = "I Am Fine"
139     * </pre>
140     *
141     * @param str  The String to capitalize, may be null.
142     * @return capitalized String, {@code null} if null String input.
143     */
144    public static String capitalizeFully(final String str) {
145        return capitalizeFully(str, null);
146    }
147
148    /**
149     * Converts all the delimiter separated words in a String into capitalized words,
150     * that is each word is made up of a titlecase character and then a series of
151     * lowercase characters.
152     *
153     * <p>
154     * The delimiters represent a set of characters understood to separate words.
155     * The first string character and the first non-delimiter character after a
156     * delimiter will be capitalized.
157     * </p>
158     *
159     * <p>
160     * A {@code null} input String returns {@code null}.
161     * Capitalization uses the Unicode title case, normally equivalent to
162     * upper case.
163     * </p>
164     *
165     * <pre>
166     * WordUtils.capitalizeFully(null, *)            = null
167     * WordUtils.capitalizeFully("", *)              = ""
168     * WordUtils.capitalizeFully(*, null)            = *
169     * WordUtils.capitalizeFully(*, new char[0])     = *
170     * WordUtils.capitalizeFully("i aM.fine", {'.'}) = "I am.Fine"
171     * </pre>
172     *
173     * @param str  The String to capitalize, may be null.
174     * @param delimiters  set of characters to determine capitalization, null means whitespace.
175     * @return capitalized String, {@code null} if null String input.
176     * @since 2.1
177     */
178    public static String capitalizeFully(final String str, final char... delimiters) {
179        final int delimLen = delimiters == null ? -1 : delimiters.length;
180        if (StringUtils.isEmpty(str) || delimLen == 0) {
181            return str;
182        }
183        return capitalize(str.toLowerCase(), delimiters);
184    }
185
186    /**
187     * Checks if the String contains all words in the given array.
188     *
189     * <p>
190     * A {@code null} String will return {@code false}. A {@code null}, zero
191     * length search array or if one element of array is null will return {@code false}.
192     * </p>
193     *
194     * <pre>
195     * WordUtils.containsAllWords(null, *)            = false
196     * WordUtils.containsAllWords("", *)              = false
197     * WordUtils.containsAllWords(*, null)            = false
198     * WordUtils.containsAllWords(*, [])              = false
199     * WordUtils.containsAllWords("abcd", "ab", "cd") = false
200     * WordUtils.containsAllWords("abc def", "def", "abc") = true
201     * </pre>
202     *
203     * @param word The CharSequence to check, may be null.
204     * @param words The array of String words to search for, may be null.
205     * @return {@code true} if all search words are found, {@code false} otherwise.
206     * @since 3.5
207     */
208    public static boolean containsAllWords(final CharSequence word, final CharSequence... words) {
209        if (StringUtils.isEmpty(word) || ArrayUtils.isEmpty(words)) {
210            return false;
211        }
212        for (final CharSequence w : words) {
213            if (StringUtils.isBlank(w)) {
214                return false;
215            }
216            final Pattern p = Pattern.compile(".*\\b" + Pattern.quote(w.toString()) + "\\b.*", Pattern.DOTALL);
217            if (!p.matcher(word).matches()) {
218                return false;
219            }
220        }
221        return true;
222    }
223
224    /**
225     * Extracts the initial characters from each word in the String.
226     *
227     * <p>
228     * All first characters after whitespace are returned as a new string.
229     * Their case is not changed.
230     * </p>
231     *
232     * <p>
233     * Whitespace is defined by {@link Character#isWhitespace(char)}.
234     * A {@code null} input String returns {@code null}.
235     * </p>
236     *
237     * <pre>
238     * WordUtils.initials(null)             = null
239     * WordUtils.initials("")               = ""
240     * WordUtils.initials("Ben John Lee")   = "BJL"
241     * WordUtils.initials("Ben J.Lee")      = "BJ"
242     * </pre>
243     *
244     * @param str  The String to get initials from, may be null.
245     * @return String of initial letters, {@code null} if null String input.
246     * @see #initials(String,char[])
247     * @since 2.2
248     */
249    public static String initials(final String str) {
250        return initials(str, null);
251    }
252
253    /**
254     * Extracts the initial characters from each word in the String.
255     *
256     * <p>
257     * All first characters after the defined delimiters are returned as a new string.
258     * Their case is not changed.
259     * </p>
260     *
261     * <p>
262     * If the delimiters array is null, then Whitespace is used.
263     * Whitespace is defined by {@link Character#isWhitespace(char)}.
264     * A {@code null} input String returns {@code null}.
265     * An empty delimiter array returns an empty String.
266     * </p>
267     *
268     * <pre>
269     * WordUtils.initials(null, *)                = null
270     * WordUtils.initials("", *)                  = ""
271     * WordUtils.initials("Ben John Lee", null)   = "BJL"
272     * WordUtils.initials("Ben J.Lee", null)      = "BJ"
273     * WordUtils.initials("Ben J.Lee", [' ','.']) = "BJL"
274     * WordUtils.initials(*, new char[0])         = ""
275     * </pre>
276     *
277     * @param str  The String to get initials from, may be null.
278     * @param delimiters  set of characters to determine words, null means whitespace.
279     * @return String of initial characters, {@code null} if null String input.
280     * @see #initials(String)
281     * @since 2.2
282     */
283    public static String initials(final String str, final char... delimiters) {
284        if (StringUtils.isEmpty(str)) {
285            return str;
286        }
287        if (delimiters != null && delimiters.length == 0) {
288            return StringUtils.EMPTY;
289        }
290        final int strLen = str.length();
291        final char[] buf = new char[strLen];
292        int count = 0;
293        boolean lastWasGap = true;
294        for (int i = 0; i < strLen; i++) {
295            final char ch = str.charAt(i);
296            if (isDelimiter(ch, delimiters)) {
297                lastWasGap = true;
298                continue;  // ignore ch
299            }
300            if (lastWasGap) {
301                buf[count++] = ch;
302                // keep a supplementary code point's low surrogate with its high half
303                if (Character.isHighSurrogate(ch) && i + 1 < strLen && Character.isLowSurrogate(str.charAt(i + 1))) {
304                    buf[count++] = str.charAt(++i);
305                }
306                lastWasGap = false;
307            }
308        }
309        return new String(buf, 0, count);
310    }
311
312    /**
313     * Tests if the character is a delimiter.
314     *
315     * @param ch  The character to check.
316     * @param delimiters  The delimiters.
317     * @return true if it is a delimiter.
318     */
319    private static boolean isDelimiter(final char ch, final char[] delimiters) {
320        return delimiters == null ? Character.isWhitespace(ch) : ArrayUtils.contains(delimiters, ch);
321    }
322
323    /**
324     * Tests if the code point is a delimiter.
325     *
326     * <p>
327     * A {@code null} {@code delimiters} array treats any whitespace code point, as defined by
328     * {@link Character#isWhitespace(int)}, as a delimiter.
329     * </p>
330     *
331     * @param codePoint  The code point to check.
332     * @param delimiters  The delimiters, {@code null} matches whitespace.
333     * @return true if it is a delimiter.
334     */
335    private static boolean isDelimiter(final int codePoint, final char[] delimiters) {
336        if (delimiters == null) {
337            return Character.isWhitespace(codePoint);
338        }
339        for (final char delimiter : delimiters) {
340            if (codePoint == delimiter) {
341                return true;
342            }
343        }
344        return false;
345    }
346
347    /**
348     * Swaps the case of a String using a word based algorithm.
349     *
350     * <ul>
351     *  <li>Upper case character converts to Lower case</li>
352     *  <li>Title case character converts to Lower case</li>
353     *  <li>Lower case character after Whitespace or at start converts to Title case</li>
354     *  <li>Other Lower case character converts to Upper case</li>
355     * </ul>
356     *
357     * <p>
358     * Whitespace is defined by {@link Character#isWhitespace(char)}.
359     * A {@code null} input String returns {@code null}.
360     * </p>
361     *
362     * <pre>
363     * StringUtils.swapCase(null)                 = null
364     * StringUtils.swapCase("")                   = ""
365     * StringUtils.swapCase("The dog has a BONE") = "tHE DOG HAS A bone"
366     * </pre>
367     *
368     * @param str  The String to swap case, may be null.
369     * @return A new String, {@code null} if null String input.
370     */
371    public static String swapCase(final String str) {
372        if (StringUtils.isEmpty(str)) {
373            return str;
374        }
375        final char[] buffer = str.toCharArray();
376
377        boolean whitespace = true;
378
379        for (int i = 0; i < buffer.length;) {
380            final int codePoint = Character.codePointAt(buffer, i);
381            if (Character.isUpperCase(codePoint) || Character.isTitleCase(codePoint)) {
382                Character.toChars(Character.toLowerCase(codePoint), buffer, i);
383                whitespace = false;
384            } else if (Character.isLowerCase(codePoint)) {
385                if (whitespace) {
386                    Character.toChars(Character.toTitleCase(codePoint), buffer, i);
387                    whitespace = false;
388                } else {
389                    Character.toChars(Character.toUpperCase(codePoint), buffer, i);
390                }
391            } else {
392                whitespace = Character.isWhitespace(codePoint);
393            }
394            i += Character.charCount(codePoint);
395        }
396        return new String(buffer);
397    }
398
399    /**
400     * Uncapitalizes all the whitespace separated words in a String.
401     * Only the first character of each word is changed.
402     *
403     * <p>
404     * Whitespace is defined by {@link Character#isWhitespace(char)}.
405     * A {@code null} input String returns {@code null}.
406     * </p>
407     *
408     * <pre>
409     * WordUtils.uncapitalize(null)        = null
410     * WordUtils.uncapitalize("")          = ""
411     * WordUtils.uncapitalize("I Am FINE") = "i am fINE"
412     * </pre>
413     *
414     * @param str  The String to uncapitalize, may be null.
415     * @return uncapitalized String, {@code null} if null String input.
416     * @see #capitalize(String)
417     */
418    public static String uncapitalize(final String str) {
419        return uncapitalize(str, null);
420    }
421
422    /**
423     * Uncapitalizes all the whitespace separated words in a String.
424     * Only the first character of each word is changed.
425     *
426     * <p>
427     * The delimiters represent a set of characters understood to separate words.
428     * The first string character and the first non-delimiter character after a
429     * delimiter will be uncapitalized.
430     * </p>
431     *
432     * <p>
433     * Whitespace is defined by {@link Character#isWhitespace(char)}.
434     * A {@code null} input String returns {@code null}.
435     * </p>
436     *
437     * <pre>
438     * WordUtils.uncapitalize(null, *)            = null
439     * WordUtils.uncapitalize("", *)              = ""
440     * WordUtils.uncapitalize(*, null)            = *
441     * WordUtils.uncapitalize(*, new char[0])     = *
442     * WordUtils.uncapitalize("I AM.FINE", {'.'}) = "i AM.fINE"
443     * </pre>
444     *
445     * @param str  The String to uncapitalize, may be null.
446     * @param delimiters  set of characters to determine uncapitalization, null means whitespace.
447     * @return uncapitalized String, {@code null} if null String input.
448     * @see #capitalize(String)
449     * @since 2.1
450     */
451    public static String uncapitalize(final String str, final char... delimiters) {
452        final int delimLen = delimiters == null ? -1 : delimiters.length;
453        if (StringUtils.isEmpty(str) || delimLen == 0) {
454            return str;
455        }
456        final char[] buffer = str.toCharArray();
457        boolean uncapitalizeNext = true;
458        for (int i = 0; i < buffer.length;) {
459            final int codePoint = Character.codePointAt(buffer, i);
460            if (isDelimiter(codePoint, delimiters)) {
461                uncapitalizeNext = true;
462            } else if (uncapitalizeNext) {
463                Character.toChars(Character.toLowerCase(codePoint), buffer, i);
464                uncapitalizeNext = false;
465            }
466            i += Character.charCount(codePoint);
467        }
468        return new String(buffer);
469    }
470
471    /**
472     * Wraps a single line of text, identifying words by {@code ' '}.
473     *
474     * <p>
475     * New lines will be separated by the system property line separator.
476     * Very long words, such as URLs will <em>not</em> be wrapped.
477     * </p>
478     *
479     * <p>
480     * Leading spaces on a new line are stripped.
481     * Trailing spaces are not stripped.
482     * </p>
483     *
484     * <table border="1">
485     *  <caption>Examples</caption>
486     *  <tr>
487     *   <th>input</th>
488     *   <th>wrapLength</th>
489     *   <th>result</th>
490     *  </tr>
491     *  <tr>
492     *   <td>null</td>
493     *   <td>*</td>
494     *   <td>null</td>
495     *  </tr>
496     *  <tr>
497     *   <td>""</td>
498     *   <td>*</td>
499     *   <td>""</td>
500     *  </tr>
501     *  <tr>
502     *   <td>"Here is one line of text that is going to be wrapped after 20 columns."</td>
503     *   <td>20</td>
504     *   <td>"Here is one line of\ntext that is going\nto be wrapped after\n20 columns."</td>
505     *  </tr>
506     *  <tr>
507     *   <td>"Click here to jump to the commons website - https://commons.apache.org"</td>
508     *   <td>20</td>
509     *   <td>"Click here to jump\nto the commons\nwebsite -\nhttps://commons.apache.org"</td>
510     *  </tr>
511     *  <tr>
512     *   <td>"Click here, https://commons.apache.org, to jump to the commons website"</td>
513     *   <td>20</td>
514     *   <td>"Click here,\nhttps://commons.apache.org,\nto jump to the\ncommons website"</td>
515     *  </tr>
516     * </table>
517     *
518     * (assuming that '\n' is the systems line separator)
519     *
520     * @param str  The String to be word wrapped, may be null.
521     * @param wrapLength  The column to wrap the words at, less than 1 is treated as 1.
522     * @return A line with newlines inserted, {@code null} if null input.
523     */
524    public static String wrap(final String str, final int wrapLength) {
525        return wrap(str, wrapLength, null, false);
526    }
527
528    /**
529     * Wraps a single line of text, identifying words by {@code ' '}.
530     *
531     * <p>
532     * Leading spaces on a new line are stripped.
533     * Trailing spaces are not stripped.
534     * </p>
535     *
536     * <table border="1">
537     *  <caption>Examples</caption>
538     *  <tr>
539     *   <th>input</th>
540     *   <th>wrapLength</th>
541     *   <th>newLineString</th>
542     *   <th>wrapLongWords</th>
543     *   <th>result</th>
544     *  </tr>
545     *  <tr>
546     *   <td>null</td>
547     *   <td>*</td>
548     *   <td>*</td>
549     *   <td>true/false</td>
550     *   <td>null</td>
551     *  </tr>
552     *  <tr>
553     *   <td>""</td>
554     *   <td>*</td>
555     *   <td>*</td>
556     *   <td>true/false</td>
557     *   <td>""</td>
558     *  </tr>
559     *  <tr>
560     *   <td>"Here is one line of text that is going to be wrapped after 20 columns."</td>
561     *   <td>20</td>
562     *   <td>"\n"</td>
563     *   <td>true/false</td>
564     *   <td>"Here is one line of\ntext that is going\nto be wrapped after\n20 columns."</td>
565     *  </tr>
566     *  <tr>
567     *   <td>"Here is one line of text that is going to be wrapped after 20 columns."</td>
568     *   <td>20</td>
569     *   <td>"&lt;br /&gt;"</td>
570     *   <td>true/false</td>
571     *   <td>"Here is one line of&lt;br /&gt;text that is going&lt;br /&gt;to be wrapped after&lt;br /&gt;20 columns."</td>
572     *  </tr>
573     *  <tr>
574     *   <td>"Here is one line of text that is going to be wrapped after 20 columns."</td>
575     *   <td>20</td>
576     *   <td>null</td>
577     *   <td>true/false</td>
578     *   <td>"Here is one line of" + systemNewLine + "text that is going" + systemNewLine + "to be wrapped after" + systemNewLine + "20 columns."</td>
579     *  </tr>
580     *  <tr>
581     *   <td>"Click here to jump to the commons website - https://commons.apache.org"</td>
582     *   <td>20</td>
583     *   <td>"\n"</td>
584     *   <td>false</td>
585     *   <td>"Click here to jump\nto the commons\nwebsite -\nhttps://commons.apache.org"</td>
586     *  </tr>
587     *  <tr>
588     *   <td>"Click here to jump to the commons website - https://commons.apache.org"</td>
589     *   <td>20</td>
590     *   <td>"\n"</td>
591     *   <td>true</td>
592     *   <td>"Click here to jump\nto the commons\nwebsite -\nhttps://commons.apach\ne.org"</td>
593     *  </tr>
594     * </table>
595     *
596     * @param str  The String to be word wrapped, may be null.
597     * @param wrapLength  The column to wrap the words at, less than 1 is treated as 1.
598     * @param newLineStr  The string to insert for a new line,
599     *  {@code null} uses the system property line separator.
600     * @param wrapLongWords  true if long words (such as URLs) should be wrapped.
601     * @return A line with newlines inserted, {@code null} if null input.
602     */
603    public static String wrap(final String str, final int wrapLength, final String newLineStr, final boolean wrapLongWords) {
604        return wrap(str, wrapLength, newLineStr, wrapLongWords, " ");
605    }
606
607    /**
608     * Wraps a single line of text, identifying words by {@code wrapOn}.
609     *
610     * <p>
611     * Leading spaces on a new line are stripped.
612     * Trailing spaces are not stripped.
613     * </p>
614     *
615     * <table border="1">
616     *  <caption>Examples</caption>
617     *  <tr>
618     *   <th>input</th>
619     *   <th>wrapLength</th>
620     *   <th>newLineString</th>
621     *   <th>wrapLongWords</th>
622     *   <th>wrapOn</th>
623     *   <th>result</th>
624     *  </tr>
625     *  <tr>
626     *   <td>null</td>
627     *   <td>*</td>
628     *   <td>*</td>
629     *   <td>true/false</td>
630     *   <td>*</td>
631     *   <td>null</td>
632     *  </tr>
633     *  <tr>
634     *   <td>""</td>
635     *   <td>*</td>
636     *   <td>*</td>
637     *   <td>true/false</td>
638     *   <td>*</td>
639     *   <td>""</td>
640     *  </tr>
641     *  <tr>
642     *   <td>"Here is one line of text that is going to be wrapped after 20 columns."</td>
643     *   <td>20</td>
644     *   <td>"\n"</td>
645     *   <td>true/false</td>
646     *   <td>" "</td>
647     *   <td>"Here is one line of\ntext that is going\nto be wrapped after\n20 columns."</td>
648     *  </tr>
649     *  <tr>
650     *   <td>"Here is one line of text that is going to be wrapped after 20 columns."</td>
651     *   <td>20</td>
652     *   <td>"&lt;br /&gt;"</td>
653     *   <td>true/false</td>
654     *   <td>" "</td>
655     *   <td>"Here is one line of&lt;br /&gt;text that is going&lt;br /&gt;to be wrapped after&lt;br /&gt;20 columns."</td>
656     *  </tr>
657     *  <tr>
658     *   <td>"Here is one line of text that is going to be wrapped after 20 columns."</td>
659     *   <td>20</td>
660     *   <td>null</td>
661     *   <td>true/false</td>
662     *   <td>" "</td>
663     *   <td>"Here is one line of" + systemNewLine + "text that is going" + systemNewLine + "to be wrapped after" + systemNewLine + "20 columns."</td>
664     *  </tr>
665     *  <tr>
666     *   <td>"Click here to jump to the commons website - https://commons.apache.org"</td>
667     *   <td>20</td>
668     *   <td>"\n"</td>
669     *   <td>false</td>
670     *   <td>" "</td>
671     *   <td>"Click here to jump\nto the commons\nwebsite -\nhttps://commons.apache.org"</td>
672     *  </tr>
673     *  <tr>
674     *   <td>"Click here to jump to the commons website - https://commons.apache.org"</td>
675     *   <td>20</td>
676     *   <td>"\n"</td>
677     *   <td>true</td>
678     *   <td>" "</td>
679     *   <td>"Click here to jump\nto the commons\nwebsite -\nhttps://commons.apach\ne.org"</td>
680     *  </tr>
681     *  <tr>
682     *   <td>"flammable/inflammable"</td>
683     *   <td>20</td>
684     *   <td>"\n"</td>
685     *   <td>true</td>
686     *   <td>"/"</td>
687     *   <td>"flammable\ninflammable"</td>
688     *  </tr>
689     * </table>
690     *
691     * @param str  The String to be word wrapped, may be null.
692     * @param wrapLength  The column to wrap the words at, less than 1 is treated as 1.
693     * @param newLineStr  The string to insert for a new line,
694     *  {@code null} uses the system property line separator.
695     * @param wrapLongWords  true if long words (such as URLs) should be wrapped.
696     * @param wrapOn regex expression to be used as a breakable characters,
697     *               if blank string is provided a space character will be used.
698     * @return A line with newlines inserted, {@code null} if null input.
699     */
700    public static String wrap(final String str, int wrapLength, String newLineStr, final boolean wrapLongWords, String wrapOn) {
701        if (str == null) {
702            return null;
703        }
704        if (newLineStr == null) {
705            newLineStr = System.lineSeparator();
706        }
707        if (wrapLength < 1) {
708            wrapLength = 1;
709        }
710        if (StringUtils.isBlank(wrapOn)) {
711            wrapOn = " ";
712        }
713        final Pattern patternToWrapOn = Pattern.compile(wrapOn);
714        final int inputLineLength = str.length();
715        int offset = 0;
716        final StringBuilder wrappedLine = new StringBuilder(inputLineLength + 32);
717
718        while (offset < inputLineLength) {
719            int spaceToWrapAt = -1;
720            int endOfWrapAt = -1;
721            Matcher matcher = patternToWrapOn.matcher(
722                str.substring(offset, Math.min((int) Math.min(Integer.MAX_VALUE, offset + wrapLength + 1L), inputLineLength)));
723            if (matcher.find()) {
724                spaceToWrapAt = matcher.start() + offset;
725                endOfWrapAt = matcher.end() + offset;
726                // Skip leading match, if it is not zero-width
727                if (spaceToWrapAt == offset && endOfWrapAt != offset) {
728                    offset = endOfWrapAt;
729                    continue;
730                }
731            }
732            // only last line without leading spaces is left
733            if (inputLineLength - offset <= wrapLength) {
734                break;
735            }
736            while (matcher.find()) {
737                spaceToWrapAt = matcher.start() + offset;
738                endOfWrapAt = matcher.end() + offset;
739            }
740            if (endOfWrapAt > offset) {
741                // normal case
742                wrappedLine.append(str, offset, spaceToWrapAt);
743                wrappedLine.append(newLineStr);
744                offset = endOfWrapAt;
745            } else // really long word or URL
746            if (wrapLongWords) {
747                // wrap really long word one line at a time, but keep a surrogate pair whole
748                int wrapAt = wrapLength + offset;
749                if (Character.isHighSurrogate(str.charAt(wrapAt - 1)) && Character.isLowSurrogate(str.charAt(wrapAt))) {
750                    wrapAt++;
751                }
752                wrappedLine.append(str, offset, wrapAt);
753                wrappedLine.append(newLineStr);
754                offset = wrapAt;
755            } else {
756                // do not wrap really long word, just extend beyond limit;
757                // match against a region of the original string rather than copying the entire
758                // unbounded remainder per output line (which is quadratic), mirroring the
759                // windowed substring used by the main loop above
760                matcher = patternToWrapOn.matcher(str);
761                matcher.region(offset + wrapLength, inputLineLength);
762                spaceToWrapAt = -1;
763                if (matcher.find()) {
764                    spaceToWrapAt = matcher.start();
765                    endOfWrapAt = matcher.end();
766                }
767
768                if (spaceToWrapAt >= 0) {
769                    wrappedLine.append(str, offset, spaceToWrapAt);
770                    wrappedLine.append(newLineStr);
771                    // at least offset + wrapLength >= offset + 1
772                    offset = endOfWrapAt;
773                } else {
774                    wrappedLine.append(str, offset, str.length());
775                    offset = inputLineLength;
776                }
777            }
778        }
779        // Whatever is left in line is short enough to just pass through
780        wrappedLine.append(str, offset, str.length());
781        return wrappedLine.toString();
782    }
783
784    /**
785     * {@link WordUtils} instances should NOT be constructed in
786     * standard programming. Instead, the class should be used as
787     * {@code WordUtils.wrap("foo bar", 20);}.
788     *
789     * <p>
790     * This constructor is public to permit tools that require a JavaBean
791     * instance to operate.
792     * </p>
793     */
794    public WordUtils() {
795    }
796
797}