001/* 002 * Licensed to the Apache Software Foundation (ASF) under one or more 003 * contributor license agreements. See the NOTICE file distributed with 004 * this work for additional information regarding copyright ownership. 005 * The ASF licenses this file to You under the Apache License, Version 2.0 006 * (the "License"); you may not use this file except in compliance with 007 * the License. You may obtain a copy of the License at 008 * 009 * https://www.apache.org/licenses/LICENSE-2.0 010 * 011 * Unless required by applicable law or agreed to in writing, software 012 * distributed under the License is distributed on an "AS IS" BASIS, 013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. 014 * See the License for the specific language governing permissions and 015 * limitations under the License. 016 */ 017package org.apache.commons.lang3.text; 018 019import java.util.regex.Matcher; 020import java.util.regex.Pattern; 021 022import org.apache.commons.lang3.ArrayUtils; 023import org.apache.commons.lang3.StringUtils; 024 025/** 026 * Operations on Strings that contain words. 027 * 028 * <p> 029 * This class tries to handle {@code null} input gracefully. 030 * An exception will not be thrown for a {@code null} input. 031 * Each method documents its behavior in more detail. 032 * </p> 033 * 034 * @since 2.0 035 * @deprecated As of <a href="https://commons.apache.org/proper/commons-lang/changes-report.html#a3.6">3.6</a>, use Apache Commons Text 036 * <a href="https://commons.apache.org/proper/commons-text/javadocs/api-release/org/apache/commons/text/WordUtils.html"> 037 * WordUtils</a>. 038 */ 039@Deprecated 040public class WordUtils { 041 042 /** 043 * Capitalizes all the whitespace separated words in a String. 044 * Only the first character of each word is changed. To convert the 045 * rest of each word to lowercase at the same time, 046 * use {@link #capitalizeFully(String)}. 047 * 048 * <p> 049 * Whitespace is defined by {@link Character#isWhitespace(char)}. 050 * A {@code null} input String returns {@code null}. 051 * Capitalization uses the Unicode title case, normally equivalent to 052 * upper case. 053 * </p> 054 * 055 * <pre> 056 * WordUtils.capitalize(null) = null 057 * WordUtils.capitalize("") = "" 058 * WordUtils.capitalize("i am FINE") = "I Am FINE" 059 * </pre> 060 * 061 * @param str The String to capitalize, may be null. 062 * @return capitalized String, {@code null} if null String input. 063 * @see #uncapitalize(String) 064 * @see #capitalizeFully(String) 065 */ 066 public static String capitalize(final String str) { 067 return capitalize(str, null); 068 } 069 070 /** 071 * Capitalizes all the delimiter separated words in a String. 072 * Only the first character of each word is changed. To convert the 073 * rest of each word to lowercase at the same time, 074 * use {@link #capitalizeFully(String, char[])}. 075 * 076 * <p> 077 * The delimiters represent a set of characters understood to separate words. 078 * The first string character and the first non-delimiter character after a 079 * delimiter will be capitalized. 080 * </p> 081 * 082 * <p> 083 * A {@code null} input String returns {@code null}. 084 * Capitalization uses the Unicode title case, normally equivalent to 085 * upper case. 086 * </p> 087 * 088 * <pre> 089 * WordUtils.capitalize(null, *) = null 090 * WordUtils.capitalize("", *) = "" 091 * WordUtils.capitalize(*, new char[0]) = * 092 * WordUtils.capitalize("i am fine", null) = "I Am Fine" 093 * WordUtils.capitalize("i aM.fine", {'.'}) = "I aM.Fine" 094 * </pre> 095 * 096 * @param str The String to capitalize, may be null. 097 * @param delimiters set of characters to determine capitalization, null means whitespace. 098 * @return capitalized String, {@code null} if null String input. 099 * @see #uncapitalize(String) 100 * @see #capitalizeFully(String) 101 * @since 2.1 102 */ 103 public static String capitalize(final String str, final char... delimiters) { 104 final int delimLen = delimiters == null ? -1 : delimiters.length; 105 if (StringUtils.isEmpty(str) || delimLen == 0) { 106 return str; 107 } 108 final char[] buffer = str.toCharArray(); 109 boolean capitalizeNext = true; 110 for (int i = 0; i < buffer.length;) { 111 final int codePoint = Character.codePointAt(buffer, i); 112 if (isDelimiter(codePoint, delimiters)) { 113 capitalizeNext = true; 114 } else if (capitalizeNext) { 115 Character.toChars(Character.toTitleCase(codePoint), buffer, i); 116 capitalizeNext = false; 117 } 118 i += Character.charCount(codePoint); 119 } 120 return new String(buffer); 121 } 122 123 /** 124 * Converts all the whitespace separated words in a String into capitalized words, 125 * that is each word is made up of a titlecase character and then a series of 126 * lowercase characters. 127 * 128 * <p> 129 * Whitespace is defined by {@link Character#isWhitespace(char)}. 130 * A {@code null} input String returns {@code null}. 131 * Capitalization uses the Unicode title case, normally equivalent to 132 * upper case. 133 * </p> 134 * 135 * <pre> 136 * WordUtils.capitalizeFully(null) = null 137 * WordUtils.capitalizeFully("") = "" 138 * WordUtils.capitalizeFully("i am FINE") = "I Am Fine" 139 * </pre> 140 * 141 * @param str The String to capitalize, may be null. 142 * @return capitalized String, {@code null} if null String input. 143 */ 144 public static String capitalizeFully(final String str) { 145 return capitalizeFully(str, null); 146 } 147 148 /** 149 * Converts all the delimiter separated words in a String into capitalized words, 150 * that is each word is made up of a titlecase character and then a series of 151 * lowercase characters. 152 * 153 * <p> 154 * The delimiters represent a set of characters understood to separate words. 155 * The first string character and the first non-delimiter character after a 156 * delimiter will be capitalized. 157 * </p> 158 * 159 * <p> 160 * A {@code null} input String returns {@code null}. 161 * Capitalization uses the Unicode title case, normally equivalent to 162 * upper case. 163 * </p> 164 * 165 * <pre> 166 * WordUtils.capitalizeFully(null, *) = null 167 * WordUtils.capitalizeFully("", *) = "" 168 * WordUtils.capitalizeFully(*, null) = * 169 * WordUtils.capitalizeFully(*, new char[0]) = * 170 * WordUtils.capitalizeFully("i aM.fine", {'.'}) = "I am.Fine" 171 * </pre> 172 * 173 * @param str The String to capitalize, may be null. 174 * @param delimiters set of characters to determine capitalization, null means whitespace. 175 * @return capitalized String, {@code null} if null String input. 176 * @since 2.1 177 */ 178 public static String capitalizeFully(final String str, final char... delimiters) { 179 final int delimLen = delimiters == null ? -1 : delimiters.length; 180 if (StringUtils.isEmpty(str) || delimLen == 0) { 181 return str; 182 } 183 return capitalize(str.toLowerCase(), delimiters); 184 } 185 186 /** 187 * Checks if the String contains all words in the given array. 188 * 189 * <p> 190 * A {@code null} String will return {@code false}. A {@code null}, zero 191 * length search array or if one element of array is null will return {@code false}. 192 * </p> 193 * 194 * <pre> 195 * WordUtils.containsAllWords(null, *) = false 196 * WordUtils.containsAllWords("", *) = false 197 * WordUtils.containsAllWords(*, null) = false 198 * WordUtils.containsAllWords(*, []) = false 199 * WordUtils.containsAllWords("abcd", "ab", "cd") = false 200 * WordUtils.containsAllWords("abc def", "def", "abc") = true 201 * </pre> 202 * 203 * @param word The CharSequence to check, may be null. 204 * @param words The array of String words to search for, may be null. 205 * @return {@code true} if all search words are found, {@code false} otherwise. 206 * @since 3.5 207 */ 208 public static boolean containsAllWords(final CharSequence word, final CharSequence... words) { 209 if (StringUtils.isEmpty(word) || ArrayUtils.isEmpty(words)) { 210 return false; 211 } 212 for (final CharSequence w : words) { 213 if (StringUtils.isBlank(w)) { 214 return false; 215 } 216 final Pattern p = Pattern.compile(".*\\b" + Pattern.quote(w.toString()) + "\\b.*", Pattern.DOTALL); 217 if (!p.matcher(word).matches()) { 218 return false; 219 } 220 } 221 return true; 222 } 223 224 /** 225 * Extracts the initial characters from each word in the String. 226 * 227 * <p> 228 * All first characters after whitespace are returned as a new string. 229 * Their case is not changed. 230 * </p> 231 * 232 * <p> 233 * Whitespace is defined by {@link Character#isWhitespace(char)}. 234 * A {@code null} input String returns {@code null}. 235 * </p> 236 * 237 * <pre> 238 * WordUtils.initials(null) = null 239 * WordUtils.initials("") = "" 240 * WordUtils.initials("Ben John Lee") = "BJL" 241 * WordUtils.initials("Ben J.Lee") = "BJ" 242 * </pre> 243 * 244 * @param str The String to get initials from, may be null. 245 * @return String of initial letters, {@code null} if null String input. 246 * @see #initials(String,char[]) 247 * @since 2.2 248 */ 249 public static String initials(final String str) { 250 return initials(str, null); 251 } 252 253 /** 254 * Extracts the initial characters from each word in the String. 255 * 256 * <p> 257 * All first characters after the defined delimiters are returned as a new string. 258 * Their case is not changed. 259 * </p> 260 * 261 * <p> 262 * If the delimiters array is null, then Whitespace is used. 263 * Whitespace is defined by {@link Character#isWhitespace(char)}. 264 * A {@code null} input String returns {@code null}. 265 * An empty delimiter array returns an empty String. 266 * </p> 267 * 268 * <pre> 269 * WordUtils.initials(null, *) = null 270 * WordUtils.initials("", *) = "" 271 * WordUtils.initials("Ben John Lee", null) = "BJL" 272 * WordUtils.initials("Ben J.Lee", null) = "BJ" 273 * WordUtils.initials("Ben J.Lee", [' ','.']) = "BJL" 274 * WordUtils.initials(*, new char[0]) = "" 275 * </pre> 276 * 277 * @param str The String to get initials from, may be null. 278 * @param delimiters set of characters to determine words, null means whitespace. 279 * @return String of initial characters, {@code null} if null String input. 280 * @see #initials(String) 281 * @since 2.2 282 */ 283 public static String initials(final String str, final char... delimiters) { 284 if (StringUtils.isEmpty(str)) { 285 return str; 286 } 287 if (delimiters != null && delimiters.length == 0) { 288 return StringUtils.EMPTY; 289 } 290 final int strLen = str.length(); 291 final char[] buf = new char[strLen]; 292 int count = 0; 293 boolean lastWasGap = true; 294 for (int i = 0; i < strLen; i++) { 295 final char ch = str.charAt(i); 296 if (isDelimiter(ch, delimiters)) { 297 lastWasGap = true; 298 continue; // ignore ch 299 } 300 if (lastWasGap) { 301 buf[count++] = ch; 302 // keep a supplementary code point's low surrogate with its high half 303 if (Character.isHighSurrogate(ch) && i + 1 < strLen && Character.isLowSurrogate(str.charAt(i + 1))) { 304 buf[count++] = str.charAt(++i); 305 } 306 lastWasGap = false; 307 } 308 } 309 return new String(buf, 0, count); 310 } 311 312 /** 313 * Tests if the character is a delimiter. 314 * 315 * @param ch The character to check. 316 * @param delimiters The delimiters. 317 * @return true if it is a delimiter. 318 */ 319 private static boolean isDelimiter(final char ch, final char[] delimiters) { 320 return delimiters == null ? Character.isWhitespace(ch) : ArrayUtils.contains(delimiters, ch); 321 } 322 323 /** 324 * Tests if the code point is a delimiter. 325 * 326 * <p> 327 * A {@code null} {@code delimiters} array treats any whitespace code point, as defined by 328 * {@link Character#isWhitespace(int)}, as a delimiter. 329 * </p> 330 * 331 * @param codePoint The code point to check. 332 * @param delimiters The delimiters, {@code null} matches whitespace. 333 * @return true if it is a delimiter. 334 */ 335 private static boolean isDelimiter(final int codePoint, final char[] delimiters) { 336 if (delimiters == null) { 337 return Character.isWhitespace(codePoint); 338 } 339 for (final char delimiter : delimiters) { 340 if (codePoint == delimiter) { 341 return true; 342 } 343 } 344 return false; 345 } 346 347 /** 348 * Swaps the case of a String using a word based algorithm. 349 * 350 * <ul> 351 * <li>Upper case character converts to Lower case</li> 352 * <li>Title case character converts to Lower case</li> 353 * <li>Lower case character after Whitespace or at start converts to Title case</li> 354 * <li>Other Lower case character converts to Upper case</li> 355 * </ul> 356 * 357 * <p> 358 * Whitespace is defined by {@link Character#isWhitespace(char)}. 359 * A {@code null} input String returns {@code null}. 360 * </p> 361 * 362 * <pre> 363 * StringUtils.swapCase(null) = null 364 * StringUtils.swapCase("") = "" 365 * StringUtils.swapCase("The dog has a BONE") = "tHE DOG HAS A bone" 366 * </pre> 367 * 368 * @param str The String to swap case, may be null. 369 * @return A new String, {@code null} if null String input. 370 */ 371 public static String swapCase(final String str) { 372 if (StringUtils.isEmpty(str)) { 373 return str; 374 } 375 final char[] buffer = str.toCharArray(); 376 377 boolean whitespace = true; 378 379 for (int i = 0; i < buffer.length;) { 380 final int codePoint = Character.codePointAt(buffer, i); 381 if (Character.isUpperCase(codePoint) || Character.isTitleCase(codePoint)) { 382 Character.toChars(Character.toLowerCase(codePoint), buffer, i); 383 whitespace = false; 384 } else if (Character.isLowerCase(codePoint)) { 385 if (whitespace) { 386 Character.toChars(Character.toTitleCase(codePoint), buffer, i); 387 whitespace = false; 388 } else { 389 Character.toChars(Character.toUpperCase(codePoint), buffer, i); 390 } 391 } else { 392 whitespace = Character.isWhitespace(codePoint); 393 } 394 i += Character.charCount(codePoint); 395 } 396 return new String(buffer); 397 } 398 399 /** 400 * Uncapitalizes all the whitespace separated words in a String. 401 * Only the first character of each word is changed. 402 * 403 * <p> 404 * Whitespace is defined by {@link Character#isWhitespace(char)}. 405 * A {@code null} input String returns {@code null}. 406 * </p> 407 * 408 * <pre> 409 * WordUtils.uncapitalize(null) = null 410 * WordUtils.uncapitalize("") = "" 411 * WordUtils.uncapitalize("I Am FINE") = "i am fINE" 412 * </pre> 413 * 414 * @param str The String to uncapitalize, may be null. 415 * @return uncapitalized String, {@code null} if null String input. 416 * @see #capitalize(String) 417 */ 418 public static String uncapitalize(final String str) { 419 return uncapitalize(str, null); 420 } 421 422 /** 423 * Uncapitalizes all the whitespace separated words in a String. 424 * Only the first character of each word is changed. 425 * 426 * <p> 427 * The delimiters represent a set of characters understood to separate words. 428 * The first string character and the first non-delimiter character after a 429 * delimiter will be uncapitalized. 430 * </p> 431 * 432 * <p> 433 * Whitespace is defined by {@link Character#isWhitespace(char)}. 434 * A {@code null} input String returns {@code null}. 435 * </p> 436 * 437 * <pre> 438 * WordUtils.uncapitalize(null, *) = null 439 * WordUtils.uncapitalize("", *) = "" 440 * WordUtils.uncapitalize(*, null) = * 441 * WordUtils.uncapitalize(*, new char[0]) = * 442 * WordUtils.uncapitalize("I AM.FINE", {'.'}) = "i AM.fINE" 443 * </pre> 444 * 445 * @param str The String to uncapitalize, may be null. 446 * @param delimiters set of characters to determine uncapitalization, null means whitespace. 447 * @return uncapitalized String, {@code null} if null String input. 448 * @see #capitalize(String) 449 * @since 2.1 450 */ 451 public static String uncapitalize(final String str, final char... delimiters) { 452 final int delimLen = delimiters == null ? -1 : delimiters.length; 453 if (StringUtils.isEmpty(str) || delimLen == 0) { 454 return str; 455 } 456 final char[] buffer = str.toCharArray(); 457 boolean uncapitalizeNext = true; 458 for (int i = 0; i < buffer.length;) { 459 final int codePoint = Character.codePointAt(buffer, i); 460 if (isDelimiter(codePoint, delimiters)) { 461 uncapitalizeNext = true; 462 } else if (uncapitalizeNext) { 463 Character.toChars(Character.toLowerCase(codePoint), buffer, i); 464 uncapitalizeNext = false; 465 } 466 i += Character.charCount(codePoint); 467 } 468 return new String(buffer); 469 } 470 471 /** 472 * Wraps a single line of text, identifying words by {@code ' '}. 473 * 474 * <p> 475 * New lines will be separated by the system property line separator. 476 * Very long words, such as URLs will <em>not</em> be wrapped. 477 * </p> 478 * 479 * <p> 480 * Leading spaces on a new line are stripped. 481 * Trailing spaces are not stripped. 482 * </p> 483 * 484 * <table border="1"> 485 * <caption>Examples</caption> 486 * <tr> 487 * <th>input</th> 488 * <th>wrapLength</th> 489 * <th>result</th> 490 * </tr> 491 * <tr> 492 * <td>null</td> 493 * <td>*</td> 494 * <td>null</td> 495 * </tr> 496 * <tr> 497 * <td>""</td> 498 * <td>*</td> 499 * <td>""</td> 500 * </tr> 501 * <tr> 502 * <td>"Here is one line of text that is going to be wrapped after 20 columns."</td> 503 * <td>20</td> 504 * <td>"Here is one line of\ntext that is going\nto be wrapped after\n20 columns."</td> 505 * </tr> 506 * <tr> 507 * <td>"Click here to jump to the commons website - https://commons.apache.org"</td> 508 * <td>20</td> 509 * <td>"Click here to jump\nto the commons\nwebsite -\nhttps://commons.apache.org"</td> 510 * </tr> 511 * <tr> 512 * <td>"Click here, https://commons.apache.org, to jump to the commons website"</td> 513 * <td>20</td> 514 * <td>"Click here,\nhttps://commons.apache.org,\nto jump to the\ncommons website"</td> 515 * </tr> 516 * </table> 517 * 518 * (assuming that '\n' is the systems line separator) 519 * 520 * @param str The String to be word wrapped, may be null. 521 * @param wrapLength The column to wrap the words at, less than 1 is treated as 1. 522 * @return A line with newlines inserted, {@code null} if null input. 523 */ 524 public static String wrap(final String str, final int wrapLength) { 525 return wrap(str, wrapLength, null, false); 526 } 527 528 /** 529 * Wraps a single line of text, identifying words by {@code ' '}. 530 * 531 * <p> 532 * Leading spaces on a new line are stripped. 533 * Trailing spaces are not stripped. 534 * </p> 535 * 536 * <table border="1"> 537 * <caption>Examples</caption> 538 * <tr> 539 * <th>input</th> 540 * <th>wrapLength</th> 541 * <th>newLineString</th> 542 * <th>wrapLongWords</th> 543 * <th>result</th> 544 * </tr> 545 * <tr> 546 * <td>null</td> 547 * <td>*</td> 548 * <td>*</td> 549 * <td>true/false</td> 550 * <td>null</td> 551 * </tr> 552 * <tr> 553 * <td>""</td> 554 * <td>*</td> 555 * <td>*</td> 556 * <td>true/false</td> 557 * <td>""</td> 558 * </tr> 559 * <tr> 560 * <td>"Here is one line of text that is going to be wrapped after 20 columns."</td> 561 * <td>20</td> 562 * <td>"\n"</td> 563 * <td>true/false</td> 564 * <td>"Here is one line of\ntext that is going\nto be wrapped after\n20 columns."</td> 565 * </tr> 566 * <tr> 567 * <td>"Here is one line of text that is going to be wrapped after 20 columns."</td> 568 * <td>20</td> 569 * <td>"<br />"</td> 570 * <td>true/false</td> 571 * <td>"Here is one line of<br />text that is going<br />to be wrapped after<br />20 columns."</td> 572 * </tr> 573 * <tr> 574 * <td>"Here is one line of text that is going to be wrapped after 20 columns."</td> 575 * <td>20</td> 576 * <td>null</td> 577 * <td>true/false</td> 578 * <td>"Here is one line of" + systemNewLine + "text that is going" + systemNewLine + "to be wrapped after" + systemNewLine + "20 columns."</td> 579 * </tr> 580 * <tr> 581 * <td>"Click here to jump to the commons website - https://commons.apache.org"</td> 582 * <td>20</td> 583 * <td>"\n"</td> 584 * <td>false</td> 585 * <td>"Click here to jump\nto the commons\nwebsite -\nhttps://commons.apache.org"</td> 586 * </tr> 587 * <tr> 588 * <td>"Click here to jump to the commons website - https://commons.apache.org"</td> 589 * <td>20</td> 590 * <td>"\n"</td> 591 * <td>true</td> 592 * <td>"Click here to jump\nto the commons\nwebsite -\nhttps://commons.apach\ne.org"</td> 593 * </tr> 594 * </table> 595 * 596 * @param str The String to be word wrapped, may be null. 597 * @param wrapLength The column to wrap the words at, less than 1 is treated as 1. 598 * @param newLineStr The string to insert for a new line, 599 * {@code null} uses the system property line separator. 600 * @param wrapLongWords true if long words (such as URLs) should be wrapped. 601 * @return A line with newlines inserted, {@code null} if null input. 602 */ 603 public static String wrap(final String str, final int wrapLength, final String newLineStr, final boolean wrapLongWords) { 604 return wrap(str, wrapLength, newLineStr, wrapLongWords, " "); 605 } 606 607 /** 608 * Wraps a single line of text, identifying words by {@code wrapOn}. 609 * 610 * <p> 611 * Leading spaces on a new line are stripped. 612 * Trailing spaces are not stripped. 613 * </p> 614 * 615 * <table border="1"> 616 * <caption>Examples</caption> 617 * <tr> 618 * <th>input</th> 619 * <th>wrapLength</th> 620 * <th>newLineString</th> 621 * <th>wrapLongWords</th> 622 * <th>wrapOn</th> 623 * <th>result</th> 624 * </tr> 625 * <tr> 626 * <td>null</td> 627 * <td>*</td> 628 * <td>*</td> 629 * <td>true/false</td> 630 * <td>*</td> 631 * <td>null</td> 632 * </tr> 633 * <tr> 634 * <td>""</td> 635 * <td>*</td> 636 * <td>*</td> 637 * <td>true/false</td> 638 * <td>*</td> 639 * <td>""</td> 640 * </tr> 641 * <tr> 642 * <td>"Here is one line of text that is going to be wrapped after 20 columns."</td> 643 * <td>20</td> 644 * <td>"\n"</td> 645 * <td>true/false</td> 646 * <td>" "</td> 647 * <td>"Here is one line of\ntext that is going\nto be wrapped after\n20 columns."</td> 648 * </tr> 649 * <tr> 650 * <td>"Here is one line of text that is going to be wrapped after 20 columns."</td> 651 * <td>20</td> 652 * <td>"<br />"</td> 653 * <td>true/false</td> 654 * <td>" "</td> 655 * <td>"Here is one line of<br />text that is going<br />to be wrapped after<br />20 columns."</td> 656 * </tr> 657 * <tr> 658 * <td>"Here is one line of text that is going to be wrapped after 20 columns."</td> 659 * <td>20</td> 660 * <td>null</td> 661 * <td>true/false</td> 662 * <td>" "</td> 663 * <td>"Here is one line of" + systemNewLine + "text that is going" + systemNewLine + "to be wrapped after" + systemNewLine + "20 columns."</td> 664 * </tr> 665 * <tr> 666 * <td>"Click here to jump to the commons website - https://commons.apache.org"</td> 667 * <td>20</td> 668 * <td>"\n"</td> 669 * <td>false</td> 670 * <td>" "</td> 671 * <td>"Click here to jump\nto the commons\nwebsite -\nhttps://commons.apache.org"</td> 672 * </tr> 673 * <tr> 674 * <td>"Click here to jump to the commons website - https://commons.apache.org"</td> 675 * <td>20</td> 676 * <td>"\n"</td> 677 * <td>true</td> 678 * <td>" "</td> 679 * <td>"Click here to jump\nto the commons\nwebsite -\nhttps://commons.apach\ne.org"</td> 680 * </tr> 681 * <tr> 682 * <td>"flammable/inflammable"</td> 683 * <td>20</td> 684 * <td>"\n"</td> 685 * <td>true</td> 686 * <td>"/"</td> 687 * <td>"flammable\ninflammable"</td> 688 * </tr> 689 * </table> 690 * 691 * @param str The String to be word wrapped, may be null. 692 * @param wrapLength The column to wrap the words at, less than 1 is treated as 1. 693 * @param newLineStr The string to insert for a new line, 694 * {@code null} uses the system property line separator. 695 * @param wrapLongWords true if long words (such as URLs) should be wrapped. 696 * @param wrapOn regex expression to be used as a breakable characters, 697 * if blank string is provided a space character will be used. 698 * @return A line with newlines inserted, {@code null} if null input. 699 */ 700 public static String wrap(final String str, int wrapLength, String newLineStr, final boolean wrapLongWords, String wrapOn) { 701 if (str == null) { 702 return null; 703 } 704 if (newLineStr == null) { 705 newLineStr = System.lineSeparator(); 706 } 707 if (wrapLength < 1) { 708 wrapLength = 1; 709 } 710 if (StringUtils.isBlank(wrapOn)) { 711 wrapOn = " "; 712 } 713 final Pattern patternToWrapOn = Pattern.compile(wrapOn); 714 final int inputLineLength = str.length(); 715 int offset = 0; 716 final StringBuilder wrappedLine = new StringBuilder(inputLineLength + 32); 717 718 while (offset < inputLineLength) { 719 int spaceToWrapAt = -1; 720 int endOfWrapAt = -1; 721 Matcher matcher = patternToWrapOn.matcher( 722 str.substring(offset, Math.min((int) Math.min(Integer.MAX_VALUE, offset + wrapLength + 1L), inputLineLength))); 723 if (matcher.find()) { 724 spaceToWrapAt = matcher.start() + offset; 725 endOfWrapAt = matcher.end() + offset; 726 // Skip leading match, if it is not zero-width 727 if (spaceToWrapAt == offset && endOfWrapAt != offset) { 728 offset = endOfWrapAt; 729 continue; 730 } 731 } 732 // only last line without leading spaces is left 733 if (inputLineLength - offset <= wrapLength) { 734 break; 735 } 736 while (matcher.find()) { 737 spaceToWrapAt = matcher.start() + offset; 738 endOfWrapAt = matcher.end() + offset; 739 } 740 if (endOfWrapAt > offset) { 741 // normal case 742 wrappedLine.append(str, offset, spaceToWrapAt); 743 wrappedLine.append(newLineStr); 744 offset = endOfWrapAt; 745 } else // really long word or URL 746 if (wrapLongWords) { 747 // wrap really long word one line at a time, but keep a surrogate pair whole 748 int wrapAt = wrapLength + offset; 749 if (Character.isHighSurrogate(str.charAt(wrapAt - 1)) && Character.isLowSurrogate(str.charAt(wrapAt))) { 750 wrapAt++; 751 } 752 wrappedLine.append(str, offset, wrapAt); 753 wrappedLine.append(newLineStr); 754 offset = wrapAt; 755 } else { 756 // do not wrap really long word, just extend beyond limit; 757 // match against a region of the original string rather than copying the entire 758 // unbounded remainder per output line (which is quadratic), mirroring the 759 // windowed substring used by the main loop above 760 matcher = patternToWrapOn.matcher(str); 761 matcher.region(offset + wrapLength, inputLineLength); 762 spaceToWrapAt = -1; 763 if (matcher.find()) { 764 spaceToWrapAt = matcher.start(); 765 endOfWrapAt = matcher.end(); 766 } 767 768 if (spaceToWrapAt >= 0) { 769 wrappedLine.append(str, offset, spaceToWrapAt); 770 wrappedLine.append(newLineStr); 771 // at least offset + wrapLength >= offset + 1 772 offset = endOfWrapAt; 773 } else { 774 wrappedLine.append(str, offset, str.length()); 775 offset = inputLineLength; 776 } 777 } 778 } 779 // Whatever is left in line is short enough to just pass through 780 wrappedLine.append(str, offset, str.length()); 781 return wrappedLine.toString(); 782 } 783 784 /** 785 * {@link WordUtils} instances should NOT be constructed in 786 * standard programming. Instead, the class should be used as 787 * {@code WordUtils.wrap("foo bar", 20);}. 788 * 789 * <p> 790 * This constructor is public to permit tools that require a JavaBean 791 * instance to operate. 792 * </p> 793 */ 794 public WordUtils() { 795 } 796 797}