001/* 002 * Licensed to the Apache Software Foundation (ASF) under one or more 003 * contributor license agreements. See the NOTICE file distributed with 004 * this work for additional information regarding copyright ownership. 005 * The ASF licenses this file to You under the Apache License, Version 2.0 006 * (the "License"); you may not use this file except in compliance with 007 * the License. You may obtain a copy of the License at 008 * 009 * https://www.apache.org/licenses/LICENSE-2.0 010 * 011 * Unless required by applicable law or agreed to in writing, software 012 * distributed under the License is distributed on an "AS IS" BASIS, 013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. 014 * See the License for the specific language governing permissions and 015 * limitations under the License. 016 */ 017package org.apache.commons.lang3.text; 018 019import java.util.ArrayList; 020import java.util.Arrays; 021import java.util.Collections; 022import java.util.List; 023import java.util.ListIterator; 024import java.util.NoSuchElementException; 025import java.util.StringTokenizer; 026 027import org.apache.commons.lang3.ArrayUtils; 028import org.apache.commons.lang3.StringUtils; 029 030/** 031 * Tokenizes a string based on delimiters (separators) 032 * and supporting quoting and ignored character concepts. 033 * <p> 034 * This class can split a String into many smaller strings. It aims 035 * to do a similar job to {@link java.util.StringTokenizer StringTokenizer}, 036 * however it offers much more control and flexibility including implementing 037 * the {@link ListIterator} interface. By default, it is set up 038 * like {@link StringTokenizer}. 039 * </p> 040 * <p> 041 * The input String is split into a number of <em>tokens</em>. 042 * Each token is separated from the next String by a <em>delimiter</em>. 043 * One or more delimiter characters must be specified. 044 * </p> 045 * <p> 046 * Each token may be surrounded by quotes. 047 * The <em>quote</em> matcher specifies the quote character(s). 048 * A quote may be escaped within a quoted section by duplicating itself. 049 * </p> 050 * <p> 051 * Between each token and the delimiter are potentially characters that need trimming. 052 * The <em>trimmer</em> matcher specifies these characters. 053 * One usage might be to trim whitespace characters. 054 * </p> 055 * <p> 056 * At any point outside the quotes there might potentially be invalid characters. 057 * The <em>ignored</em> matcher specifies these characters to be removed. 058 * One usage might be to remove new line characters. 059 * </p> 060 * <p> 061 * Empty tokens may be removed or returned as null. 062 * </p> 063 * <pre> 064 * "a,b,c" - Three tokens "a","b","c" (comma delimiter) 065 * " a, b , c " - Three tokens "a","b","c" (default CSV processing trims whitespace) 066 * "a, ", b ,", c" - Three tokens "a, " , " b ", ", c" (quoted text untouched) 067 * </pre> 068 * 069 * <table> 070 * <caption>StrTokenizer properties and options</caption> 071 * <tr> 072 * <th>Property</th><th>Type</th><th>Default</th> 073 * </tr> 074 * <tr> 075 * <td>delim</td><td>CharSetMatcher</td><td>{ \t\n\r\f}</td> 076 * </tr> 077 * <tr> 078 * <td>quote</td><td>NoneMatcher</td><td>{}</td> 079 * </tr> 080 * <tr> 081 * <td>ignore</td><td>NoneMatcher</td><td>{}</td> 082 * </tr> 083 * <tr> 084 * <td>emptyTokenAsNull</td><td>boolean</td><td>false</td> 085 * </tr> 086 * <tr> 087 * <td>ignoreEmptyTokens</td><td>boolean</td><td>true</td> 088 * </tr> 089 * </table> 090 * 091 * @since 2.2 092 * @deprecated As of <a href="https://commons.apache.org/proper/commons-lang/changes-report.html#a3.6">3.6</a>, use Apache Commons Text 093 * <a href="https://commons.apache.org/proper/commons-text/javadocs/api-release/org/apache/commons/text/StringTokenizer.html"> 094 * StringTokenizer</a>. 095 */ 096@Deprecated 097public class StrTokenizer implements ListIterator<String>, Cloneable { 098 099 // @formatter:off 100 private static final StrTokenizer CSV_TOKENIZER_PROTOTYPE = new StrTokenizer() 101 .setDelimiterMatcher(StrMatcher.commaMatcher()) 102 .setQuoteMatcher(StrMatcher.doubleQuoteMatcher()) 103 .setIgnoredMatcher(StrMatcher.noneMatcher()) 104 .setTrimmerMatcher(StrMatcher.trimMatcher()) 105 .setEmptyTokenAsNull(false) 106 .setIgnoreEmptyTokens(false); 107 // @formatter:on 108 109 // @formatter:off 110 private static final StrTokenizer TSV_TOKENIZER_PROTOTYPE = new StrTokenizer() 111 .setDelimiterMatcher(StrMatcher.tabMatcher()) 112 .setQuoteMatcher(StrMatcher.doubleQuoteMatcher()) 113 .setIgnoredMatcher(StrMatcher.noneMatcher()) 114 .setTrimmerMatcher(StrMatcher.trimMatcher()) 115 .setEmptyTokenAsNull(false) 116 .setIgnoreEmptyTokens(false); 117 // @formatter:on 118 119 /** 120 * Gets a clone of {@code CSV_TOKENIZER_PROTOTYPE}. 121 * 122 * @return A clone of {@code CSV_TOKENIZER_PROTOTYPE}. 123 */ 124 private static StrTokenizer getCSVClone() { 125 return (StrTokenizer) CSV_TOKENIZER_PROTOTYPE.clone(); 126 } 127 128 /** 129 * Gets a new tokenizer instance which parses Comma Separated Value strings 130 * initializing it with the given input. The default for CSV processing 131 * will be trim whitespace from both ends (which can be overridden with 132 * the setTrimmer method). 133 * <p> 134 * You must call a "reset" method to set the string which you want to parse. 135 * </p> 136 * 137 * @return A new tokenizer instance which parses Comma Separated Value strings. 138 */ 139 public static StrTokenizer getCSVInstance() { 140 return getCSVClone(); 141 } 142 143 /** 144 * Gets a new tokenizer instance which parses Comma Separated Value strings 145 * initializing it with the given input. The default for CSV processing 146 * will be trim whitespace from both ends (which can be overridden with 147 * the setTrimmer method). 148 * 149 * @param input The text to parse. 150 * @return A new tokenizer instance which parses Comma Separated Value strings. 151 */ 152 public static StrTokenizer getCSVInstance(final char[] input) { 153 final StrTokenizer tok = getCSVClone(); 154 tok.reset(input); 155 return tok; 156 } 157 158 /** 159 * Gets a new tokenizer instance which parses Comma Separated Value strings 160 * initializing it with the given input. The default for CSV processing 161 * will be trim whitespace from both ends (which can be overridden with 162 * the setTrimmer method). 163 * 164 * @param input The text to parse. 165 * @return A new tokenizer instance which parses Comma Separated Value strings. 166 */ 167 public static StrTokenizer getCSVInstance(final String input) { 168 final StrTokenizer tok = getCSVClone(); 169 tok.reset(input); 170 return tok; 171 } 172 173 /** 174 * Gets a clone of {@code TSV_TOKENIZER_PROTOTYPE}. 175 * 176 * @return A clone of {@code TSV_TOKENIZER_PROTOTYPE}. 177 */ 178 private static StrTokenizer getTSVClone() { 179 return (StrTokenizer) TSV_TOKENIZER_PROTOTYPE.clone(); 180 } 181 182 /** 183 * Gets a new tokenizer instance which parses Tab Separated Value strings. 184 * The default for CSV processing will be trim whitespace from both ends 185 * (which can be overridden with the setTrimmer method). 186 * <p> 187 * You must call a "reset" method to set the string which you want to parse. 188 * </p> 189 * 190 * @return A new tokenizer instance which parses Tab Separated Value strings. 191 */ 192 public static StrTokenizer getTSVInstance() { 193 return getTSVClone(); 194 } 195 196 /** 197 * Gets a new tokenizer instance which parses Tab Separated Value strings. 198 * The default for CSV processing will be trim whitespace from both ends 199 * (which can be overridden with the setTrimmer method). 200 * 201 * @param input The string to parse. 202 * @return A new tokenizer instance which parses Tab Separated Value strings. 203 */ 204 public static StrTokenizer getTSVInstance(final char[] input) { 205 final StrTokenizer tok = getTSVClone(); 206 tok.reset(input); 207 return tok; 208 } 209 210 /** 211 * Gets a new tokenizer instance which parses Tab Separated Value strings. 212 * The default for CSV processing will be trim whitespace from both ends 213 * (which can be overridden with the setTrimmer method). 214 * 215 * @param input The string to parse. 216 * @return A new tokenizer instance which parses Tab Separated Value strings. 217 */ 218 public static StrTokenizer getTSVInstance(final String input) { 219 final StrTokenizer tok = getTSVClone(); 220 tok.reset(input); 221 return tok; 222 } 223 224 /** The text to work on. */ 225 private char[] chars; 226 227 /** The parsed tokens */ 228 private String[] tokens; 229 230 /** The current iteration position */ 231 private int tokenPos; 232 233 /** The delimiter matcher */ 234 private StrMatcher delimMatcher = StrMatcher.splitMatcher(); 235 236 /** The quote matcher */ 237 private StrMatcher quoteMatcher = StrMatcher.noneMatcher(); 238 239 /** The ignored matcher */ 240 private StrMatcher ignoredMatcher = StrMatcher.noneMatcher(); 241 242 /** The trimmer matcher */ 243 private StrMatcher trimmerMatcher = StrMatcher.noneMatcher(); 244 245 /** Whether to return empty tokens as null */ 246 private boolean emptyAsNull; 247 248 /** Whether to ignore empty tokens */ 249 private boolean ignoreEmptyTokens = true; 250 251 /** 252 * Constructs a tokenizer splitting on space, tab, newline and formfeed 253 * as per StringTokenizer, but with no text to tokenize. 254 * <p> 255 * This constructor is normally used with {@link #reset(String)}. 256 * </p> 257 */ 258 public StrTokenizer() { 259 this.chars = null; 260 } 261 262 /** 263 * Constructs a tokenizer splitting on space, tab, newline and formfeed 264 * as per StringTokenizer. 265 * 266 * @param input The string which is to be parsed, not cloned. 267 */ 268 public StrTokenizer(final char[] input) { 269 this.chars = ArrayUtils.clone(input); 270 } 271 272 /** 273 * Constructs a tokenizer splitting on the specified character. 274 * 275 * @param input The string which is to be parsed, not cloned. 276 * @param delim The field delimiter character. 277 */ 278 public StrTokenizer(final char[] input, final char delim) { 279 this(input); 280 setDelimiterChar(delim); 281 } 282 283 /** 284 * Constructs a tokenizer splitting on the specified delimiter character 285 * and handling quotes using the specified quote character. 286 * 287 * @param input The string which is to be parsed, not cloned. 288 * @param delim The field delimiter character. 289 * @param quote The field quoted string character. 290 */ 291 public StrTokenizer(final char[] input, final char delim, final char quote) { 292 this(input, delim); 293 setQuoteChar(quote); 294 } 295 296 /** 297 * Constructs a tokenizer splitting on the specified string. 298 * 299 * @param input The string which is to be parsed, not cloned. 300 * @param delim The field delimiter string. 301 */ 302 public StrTokenizer(final char[] input, final String delim) { 303 this(input); 304 setDelimiterString(delim); 305 } 306 307 /** 308 * Constructs a tokenizer splitting using the specified delimiter matcher. 309 * 310 * @param input The string which is to be parsed, not cloned. 311 * @param delim The field delimiter matcher. 312 */ 313 public StrTokenizer(final char[] input, final StrMatcher delim) { 314 this(input); 315 setDelimiterMatcher(delim); 316 } 317 318 /** 319 * Constructs a tokenizer splitting using the specified delimiter matcher 320 * and handling quotes using the specified quote matcher. 321 * 322 * @param input The string which is to be parsed, not cloned. 323 * @param delim The field delimiter character. 324 * @param quote The field quoted string character. 325 */ 326 public StrTokenizer(final char[] input, final StrMatcher delim, final StrMatcher quote) { 327 this(input, delim); 328 setQuoteMatcher(quote); 329 } 330 331 /** 332 * Constructs a tokenizer splitting on space, tab, newline and formfeed 333 * as per StringTokenizer. 334 * 335 * @param input The string which is to be parsed. 336 */ 337 public StrTokenizer(final String input) { 338 if (input != null) { 339 chars = input.toCharArray(); 340 } else { 341 chars = null; 342 } 343 } 344 345 /** 346 * Constructs a tokenizer splitting on the specified delimiter character. 347 * 348 * @param input The string which is to be parsed. 349 * @param delim The field delimiter character. 350 */ 351 public StrTokenizer(final String input, final char delim) { 352 this(input); 353 setDelimiterChar(delim); 354 } 355 356 /** 357 * Constructs a tokenizer splitting on the specified delimiter character 358 * and handling quotes using the specified quote character. 359 * 360 * @param input The string which is to be parsed. 361 * @param delim The field delimiter character. 362 * @param quote The field quoted string character. 363 */ 364 public StrTokenizer(final String input, final char delim, final char quote) { 365 this(input, delim); 366 setQuoteChar(quote); 367 } 368 369 /** 370 * Constructs a tokenizer splitting on the specified delimiter string. 371 * 372 * @param input The string which is to be parsed. 373 * @param delim The field delimiter string. 374 */ 375 public StrTokenizer(final String input, final String delim) { 376 this(input); 377 setDelimiterString(delim); 378 } 379 380 /** 381 * Constructs a tokenizer splitting using the specified delimiter matcher. 382 * 383 * @param input The string which is to be parsed. 384 * @param delim The field delimiter matcher. 385 */ 386 public StrTokenizer(final String input, final StrMatcher delim) { 387 this(input); 388 setDelimiterMatcher(delim); 389 } 390 391 /** 392 * Constructs a tokenizer splitting using the specified delimiter matcher 393 * and handling quotes using the specified quote matcher. 394 * 395 * @param input The string which is to be parsed. 396 * @param delim The field delimiter matcher. 397 * @param quote The field quoted string matcher. 398 */ 399 public StrTokenizer(final String input, final StrMatcher delim, final StrMatcher quote) { 400 this(input, delim); 401 setQuoteMatcher(quote); 402 } 403 404 /** 405 * Always throws {@link UnsupportedOperationException}. 406 * 407 * @param obj this parameter ignored. 408 * @throws UnsupportedOperationException Thrown because this operation is unsupported. 409 */ 410 @Override 411 public void add(final String obj) { 412 throw new UnsupportedOperationException("add() is unsupported"); 413 } 414 415 /** 416 * Adds a token to a list, paying attention to the parameters we've set. 417 * 418 * @param list The list to add to. 419 * @param tok The token to add. 420 */ 421 private void addToken(final List<String> list, String tok) { 422 if (StringUtils.isEmpty(tok)) { 423 if (isIgnoreEmptyTokens()) { 424 return; 425 } 426 if (isEmptyTokenAsNull()) { 427 tok = null; 428 } 429 } 430 list.add(tok); 431 } 432 433 /** 434 * Checks if tokenization has been done, and if not then do it. 435 */ 436 private void checkTokenized() { 437 if (tokens == null) { 438 if (chars == null) { 439 // still call tokenize as subclass may do some work 440 final List<String> split = tokenize(null, 0, 0); 441 tokens = split.toArray(ArrayUtils.EMPTY_STRING_ARRAY); 442 } else { 443 final List<String> split = tokenize(chars, 0, chars.length); 444 tokens = split.toArray(ArrayUtils.EMPTY_STRING_ARRAY); 445 } 446 } 447 } 448 449 /** 450 * Creates a new instance of this Tokenizer. The new instance is reset so 451 * that it will be at the start of the token list. 452 * If a {@link CloneNotSupportedException} is caught, return {@code null}. 453 * 454 * @return A new instance of this Tokenizer which has been reset. 455 */ 456 @Override 457 public Object clone() { 458 try { 459 return cloneReset(); 460 } catch (final CloneNotSupportedException ex) { 461 return null; 462 } 463 } 464 465 /** 466 * Creates a new instance of this Tokenizer. The new instance is reset so that 467 * it will be at the start of the token list. 468 * 469 * @return A new instance of this Tokenizer which has been reset. 470 * @throws CloneNotSupportedException Thrown if there is a problem cloning. 471 */ 472 Object cloneReset() throws CloneNotSupportedException { 473 // this method exists to enable 100% test coverage 474 final StrTokenizer cloned = (StrTokenizer) super.clone(); 475 if (cloned.chars != null) { 476 cloned.chars = cloned.chars.clone(); 477 } 478 cloned.reset(); 479 return cloned; 480 } 481 482 /** 483 * Gets the String content that the tokenizer is parsing. 484 * 485 * @return The string content being parsed. 486 */ 487 public String getContent() { 488 if (chars == null) { 489 return null; 490 } 491 return new String(chars); 492 } 493 494 /** 495 * Gets the field delimiter matcher. 496 * 497 * @return The delimiter matcher in use. 498 */ 499 public StrMatcher getDelimiterMatcher() { 500 return this.delimMatcher; 501 } 502 503 /** 504 * Gets the ignored character matcher. 505 * <p> 506 * These characters are ignored when parsing the String, unless they are 507 * within a quoted region. 508 * The default value is not to ignore anything. 509 * </p> 510 * 511 * @return The ignored matcher in use. 512 */ 513 public StrMatcher getIgnoredMatcher() { 514 return ignoredMatcher; 515 } 516 517 /** 518 * Gets the quote matcher currently in use. 519 * <p> 520 * The quote character is used to wrap data between the tokens. 521 * This enables delimiters to be entered as data. 522 * The default value is '"' (double quote). 523 * </p> 524 * 525 * @return The quote matcher in use. 526 */ 527 public StrMatcher getQuoteMatcher() { 528 return quoteMatcher; 529 } 530 531 /** 532 * Gets a copy of the full token list as an independent modifiable array. 533 * 534 * @return The tokens as a String array. 535 */ 536 public String[] getTokenArray() { 537 checkTokenized(); 538 return tokens.clone(); 539 } 540 541 /** 542 * Gets a copy of the full token list as an independent modifiable list. 543 * 544 * @return The tokens as a String array. 545 */ 546 public List<String> getTokenList() { 547 checkTokenized(); 548 final List<String> list = new ArrayList<>(tokens.length); 549 list.addAll(Arrays.asList(tokens)); 550 return list; 551 } 552 553 /** 554 * Gets the trimmer character matcher. 555 * <p> 556 * These characters are trimmed off on each side of the delimiter 557 * until the token or quote is found. 558 * The default value is not to trim anything. 559 * </p> 560 * 561 * @return The trimmer matcher in use. 562 */ 563 public StrMatcher getTrimmerMatcher() { 564 return trimmerMatcher; 565 } 566 567 /** 568 * Tests whether there are any more tokens. 569 * 570 * @return true if there are more tokens. 571 */ 572 @Override 573 public boolean hasNext() { 574 checkTokenized(); 575 return tokenPos < tokens.length; 576 } 577 578 /** 579 * Tests whether there are any previous tokens that can be iterated to. 580 * 581 * @return true if there are previous tokens. 582 */ 583 @Override 584 public boolean hasPrevious() { 585 checkTokenized(); 586 return tokenPos > 0; 587 } 588 589 /** 590 * Tests whether the tokenizer currently returns empty tokens as null. The default for this property is false. 591 * 592 * @return true if empty tokens are returned as null. 593 */ 594 public boolean isEmptyTokenAsNull() { 595 return this.emptyAsNull; 596 } 597 598 /** 599 * Tests whether the tokenizer currently ignores empty tokens. The default for this property is true. 600 * 601 * @return true if empty tokens are not returned. 602 */ 603 public boolean isIgnoreEmptyTokens() { 604 return ignoreEmptyTokens; 605 } 606 607 /** 608 * Tests whether the characters at the index specified match the quote already matched in readNextToken(). 609 * 610 * @param srcChars The character array being tokenized. 611 * @param pos The position to check for a quote. 612 * @param len The length of the character array being tokenized. 613 * @param quoteStart The start position of the matched quote, 0 if no quoting. 614 * @param quoteLen The length of the matched quote, 0 if no quoting. 615 * @return true if a quote is matched. 616 */ 617 private boolean isQuote(final char[] srcChars, final int pos, final int len, final int quoteStart, final int quoteLen) { 618 for (int i = 0; i < quoteLen; i++) { 619 if (pos + i >= len || srcChars[pos + i] != srcChars[quoteStart + i]) { 620 return false; 621 } 622 } 623 return true; 624 } 625 626 /** 627 * Gets the next token. 628 * 629 * @return The next String token. 630 * @throws NoSuchElementException Thrown if there are no more elements. 631 */ 632 @Override 633 public String next() { 634 if (hasNext()) { 635 return tokens[tokenPos++]; 636 } 637 throw new NoSuchElementException(); 638 } 639 640 /** 641 * Gets the index of the next token to return. 642 * 643 * @return The next token index. 644 */ 645 @Override 646 public int nextIndex() { 647 return tokenPos; 648 } 649 650 /** 651 * Gets the next token from the String. 652 * Equivalent to {@link #next()} except it returns null rather than 653 * throwing {@link NoSuchElementException} when no tokens remain. 654 * 655 * @return The next sequential token, or null when no more tokens are found. 656 */ 657 public String nextToken() { 658 if (hasNext()) { 659 return tokens[tokenPos++]; 660 } 661 return null; 662 } 663 664 /** 665 * Gets the token previous to the last returned token. 666 * 667 * @return The previous token. 668 */ 669 @Override 670 public String previous() { 671 if (hasPrevious()) { 672 return tokens[--tokenPos]; 673 } 674 throw new NoSuchElementException(); 675 } 676 677 /** 678 * Gets the index of the previous token. 679 * 680 * @return The previous token index. 681 */ 682 @Override 683 public int previousIndex() { 684 return tokenPos - 1; 685 } 686 687 /** 688 * Gets the previous token from the String. 689 * 690 * @return The previous sequential token, or null when no more tokens are found. 691 */ 692 public String previousToken() { 693 if (hasPrevious()) { 694 return tokens[--tokenPos]; 695 } 696 return null; 697 } 698 699 /** 700 * Reads character by character through the String to get the next token. 701 * 702 * @param srcChars The character array being tokenized. 703 * @param start The first character of field. 704 * @param len The length of the character array being tokenized. 705 * @param workArea A temporary work area. 706 * @param tokenList The list of parsed tokens. 707 * @return The starting position of the next field (the character 708 * immediately after the delimiter), or -1 if end of string found. 709 */ 710 private int readNextToken(final char[] srcChars, int start, final int len, final StrBuilder workArea, final List<String> tokenList) { 711 // skip all leading whitespace, unless it is the 712 // field delimiter or the quote character 713 while (start < len) { 714 final int removeLen = Math.max( 715 getIgnoredMatcher().isMatch(srcChars, start, start, len), 716 getTrimmerMatcher().isMatch(srcChars, start, start, len)); 717 if (removeLen == 0 || 718 getDelimiterMatcher().isMatch(srcChars, start, start, len) > 0 || 719 getQuoteMatcher().isMatch(srcChars, start, start, len) > 0) { 720 break; 721 } 722 start += removeLen; 723 } 724 725 // handle reaching end 726 if (start >= len) { 727 addToken(tokenList, StringUtils.EMPTY); 728 return -1; 729 } 730 731 // handle empty token 732 final int delimLen = getDelimiterMatcher().isMatch(srcChars, start, start, len); 733 if (delimLen > 0) { 734 addToken(tokenList, StringUtils.EMPTY); 735 return start + delimLen; 736 } 737 738 // handle found token 739 final int quoteLen = getQuoteMatcher().isMatch(srcChars, start, start, len); 740 if (quoteLen > 0) { 741 return readWithQuotes(srcChars, start + quoteLen, len, workArea, tokenList, start, quoteLen); 742 } 743 return readWithQuotes(srcChars, start, len, workArea, tokenList, 0, 0); 744 } 745 746 /** 747 * Reads a possibly quoted string token. 748 * 749 * @param srcChars The character array being tokenized. 750 * @param start The first character of field. 751 * @param len The length of the character array being tokenized. 752 * @param workArea A temporary work area. 753 * @param tokenList The list of parsed tokens. 754 * @param quoteStart The start position of the matched quote, 0 if no quoting. 755 * @param quoteLen The length of the matched quote, 0 if no quoting. 756 * @return The starting position of the next field (the character 757 * immediately after the delimiter, or if end of string found, 758 * then the length of string. 759 */ 760 private int readWithQuotes(final char[] srcChars, final int start, final int len, final StrBuilder workArea, 761 final List<String> tokenList, final int quoteStart, final int quoteLen) { 762 // Loop until we've found the end of the quoted 763 // string or the end of the input 764 workArea.clear(); 765 int pos = start; 766 boolean quoting = quoteLen > 0; 767 int trimStart = 0; 768 769 while (pos < len) { 770 // quoting mode can occur several times throughout a string 771 // we must switch between quoting and non-quoting until we 772 // encounter a non-quoted delimiter, or end of string 773 if (quoting) { 774 // In quoting mode 775 776 // If we've found a quote character, see if it's 777 // followed by a second quote. If so, then we need 778 // to actually put the quote character into the token 779 // rather than end the token. 780 if (isQuote(srcChars, pos, len, quoteStart, quoteLen)) { 781 if (isQuote(srcChars, pos + quoteLen, len, quoteStart, quoteLen)) { 782 // matched pair of quotes, thus an escaped quote 783 workArea.append(srcChars, pos, quoteLen); 784 pos += quoteLen * 2; 785 trimStart = workArea.size(); 786 continue; 787 } 788 789 // end of quoting 790 quoting = false; 791 pos += quoteLen; 792 continue; 793 } 794 795 } else { 796 // Not in quoting mode 797 798 // check for delimiter, and thus end of token 799 final int delimLen = getDelimiterMatcher().isMatch(srcChars, pos, start, len); 800 if (delimLen > 0) { 801 // return condition when end of token found 802 addToken(tokenList, workArea.substring(0, trimStart)); 803 return pos + delimLen; 804 } 805 806 // check for quote, and thus back into quoting mode 807 if (quoteLen > 0 && isQuote(srcChars, pos, len, quoteStart, quoteLen)) { 808 quoting = true; 809 pos += quoteLen; 810 continue; 811 } 812 813 // check for ignored (outside quotes), and ignore 814 final int ignoredLen = getIgnoredMatcher().isMatch(srcChars, pos, start, len); 815 if (ignoredLen > 0) { 816 pos += ignoredLen; 817 continue; 818 } 819 820 // check for trimmed character 821 // don't yet know if it's at the end, so copy to workArea 822 // use trimStart to keep track of trim at the end 823 final int trimmedLen = getTrimmerMatcher().isMatch(srcChars, pos, start, len); 824 if (trimmedLen > 0) { 825 workArea.append(srcChars, pos, trimmedLen); 826 pos += trimmedLen; 827 continue; 828 } 829 } 830 // copy regular character from inside quotes 831 workArea.append(srcChars[pos++]); 832 trimStart = workArea.size(); 833 } 834 835 // return condition when end of string found 836 addToken(tokenList, workArea.substring(0, trimStart)); 837 return -1; 838 } 839 840 /** 841 * Always throws {@link UnsupportedOperationException}. 842 * 843 * @throws UnsupportedOperationException Thrown because this operation is unsupported. 844 */ 845 @Override 846 public void remove() { 847 throw new UnsupportedOperationException("remove() is unsupported"); 848 } 849 850 /** 851 * Resets this tokenizer, forgetting all parsing and iteration already completed. 852 * <p> 853 * This method allows the same tokenizer to be reused for the same String. 854 * </p> 855 * 856 * @return {@code this} instance. 857 */ 858 public StrTokenizer reset() { 859 tokenPos = 0; 860 tokens = null; 861 return this; 862 } 863 864 /** 865 * Reset this tokenizer, giving it a new input string to parse. 866 * In this manner you can re-use a tokenizer with the same settings 867 * on multiple input lines. 868 * 869 * @param input The new character array to tokenize, not cloned, null sets no text to parse. 870 * @return {@code this} instance. 871 */ 872 public StrTokenizer reset(final char[] input) { 873 reset(); 874 this.chars = ArrayUtils.clone(input); 875 return this; 876 } 877 878 /** 879 * Reset this tokenizer, giving it a new input string to parse. 880 * In this manner you can re-use a tokenizer with the same settings 881 * on multiple input lines. 882 * 883 * @param input The new string to tokenize, null sets no text to parse. 884 * @return {@code this} instance. 885 */ 886 public StrTokenizer reset(final String input) { 887 reset(); 888 if (input != null) { 889 this.chars = input.toCharArray(); 890 } else { 891 this.chars = null; 892 } 893 return this; 894 } 895 896 /** 897 * Sets no token and always throws {@link UnsupportedOperationException}. 898 * 899 * @param obj this parameter ignored. 900 * @throws UnsupportedOperationException Thrown because this operation is unsupported. 901 */ 902 @Override 903 public void set(final String obj) { 904 throw new UnsupportedOperationException("set() is unsupported"); 905 } 906 907 /** 908 * Sets the field delimiter character. 909 * 910 * @param delim The delimiter character to use. 911 * @return {@code this} instance. 912 */ 913 public StrTokenizer setDelimiterChar(final char delim) { 914 return setDelimiterMatcher(StrMatcher.charMatcher(delim)); 915 } 916 917 /** 918 * Sets the field delimiter matcher. 919 * <p> 920 * The delimiter is used to separate one token from another. 921 * </p> 922 * 923 * @param delim The delimiter matcher to use. 924 * @return {@code this} instance. 925 */ 926 public StrTokenizer setDelimiterMatcher(final StrMatcher delim) { 927 if (delim == null) { 928 this.delimMatcher = StrMatcher.noneMatcher(); 929 } else { 930 this.delimMatcher = delim; 931 } 932 return this; 933 } 934 935 /** 936 * Sets the field delimiter string. 937 * 938 * @param delim The delimiter string to use. 939 * @return {@code this} instance. 940 */ 941 public StrTokenizer setDelimiterString(final String delim) { 942 return setDelimiterMatcher(StrMatcher.stringMatcher(delim)); 943 } 944 945 /** 946 * Sets whether the tokenizer should return empty tokens as null. 947 * The default for this property is false. 948 * 949 * @param emptyAsNull whether empty tokens are returned as null. 950 * @return {@code this} instance. 951 */ 952 public StrTokenizer setEmptyTokenAsNull(final boolean emptyAsNull) { 953 this.emptyAsNull = emptyAsNull; 954 return this; 955 } 956 957 /** 958 * Sets the character to ignore. 959 * <p> 960 * This character is ignored when parsing the String, unless it is 961 * within a quoted region. 962 * 963 * @param ignored The ignored character to use. 964 * @return {@code this} instance. 965 */ 966 public StrTokenizer setIgnoredChar(final char ignored) { 967 return setIgnoredMatcher(StrMatcher.charMatcher(ignored)); 968 } 969 970 /** 971 * Sets the matcher for characters to ignore. 972 * <p> 973 * These characters are ignored when parsing the String, unless they are 974 * within a quoted region. 975 * </p> 976 * 977 * @param ignored The ignored matcher to use, null ignored. 978 * @return {@code this} instance. 979 */ 980 public StrTokenizer setIgnoredMatcher(final StrMatcher ignored) { 981 if (ignored != null) { 982 this.ignoredMatcher = ignored; 983 } 984 return this; 985 } 986 987 /** 988 * Sets whether the tokenizer should ignore and not return empty tokens. 989 * The default for this property is true. 990 * 991 * @param ignoreEmptyTokens whether empty tokens are not returned. 992 * @return {@code this} instance. 993 */ 994 public StrTokenizer setIgnoreEmptyTokens(final boolean ignoreEmptyTokens) { 995 this.ignoreEmptyTokens = ignoreEmptyTokens; 996 return this; 997 } 998 999 /** 1000 * Sets the quote character to use. 1001 * <p> 1002 * The quote character is used to wrap data between the tokens. 1003 * This enables delimiters to be entered as data. 1004 * </p> 1005 * 1006 * @param quote The quote character to use. 1007 * @return {@code this} instance. 1008 */ 1009 public StrTokenizer setQuoteChar(final char quote) { 1010 return setQuoteMatcher(StrMatcher.charMatcher(quote)); 1011 } 1012 1013 /** 1014 * Sets the quote matcher to use. 1015 * <p> 1016 * The quote character is used to wrap data between the tokens. 1017 * This enables delimiters to be entered as data. 1018 * </p> 1019 * 1020 * @param quote The quote matcher to use, null ignored. 1021 * @return {@code this} instance. 1022 */ 1023 public StrTokenizer setQuoteMatcher(final StrMatcher quote) { 1024 if (quote != null) { 1025 this.quoteMatcher = quote; 1026 } 1027 return this; 1028 } 1029 1030 /** 1031 * Sets the matcher for characters to trim. 1032 * <p> 1033 * These characters are trimmed off on each side of the delimiter 1034 * until the token or quote is found. 1035 * </p> 1036 * 1037 * @param trimmer The trimmer matcher to use, null ignored. 1038 * @return {@code this} instance. 1039 */ 1040 public StrTokenizer setTrimmerMatcher(final StrMatcher trimmer) { 1041 if (trimmer != null) { 1042 this.trimmerMatcher = trimmer; 1043 } 1044 return this; 1045 } 1046 1047 /** 1048 * Gets the number of tokens found in the String. 1049 * 1050 * @return The number of matched tokens. 1051 */ 1052 public int size() { 1053 checkTokenized(); 1054 return tokens.length; 1055 } 1056 1057 /** 1058 * Internal method to performs the tokenization. 1059 * <p> 1060 * Most users of this class do not need to call this method. This method 1061 * will be called automatically by other (public) methods when required. 1062 * </p> 1063 * <p> 1064 * This method exists to allow subclasses to add code before or after the 1065 * tokenization. For example, a subclass could alter the character array, 1066 * offset or count to be parsed, or call the tokenizer multiple times on 1067 * multiple strings. It is also be possible to filter the results. 1068 * </p> 1069 * <p> 1070 * {@link StrTokenizer} will always pass a zero offset and a count 1071 * equal to the length of the array to this method, however a subclass 1072 * may pass other values, or even an entirely different array. 1073 * </p> 1074 * 1075 * @param srcChars The character array being tokenized, may be null. 1076 * @param offset The start position within the character array, must be valid. 1077 * @param count The number of characters to tokenize, must be valid. 1078 * @return The modifiable list of String tokens, unmodifiable if null array or zero count. 1079 */ 1080 protected List<String> tokenize(final char[] srcChars, final int offset, final int count) { 1081 if (ArrayUtils.isEmpty(srcChars)) { 1082 return Collections.emptyList(); 1083 } 1084 final StrBuilder buf = new StrBuilder(); 1085 final List<String> tokenList = new ArrayList<>(); 1086 int pos = offset; 1087 1088 // loop around the entire buffer 1089 while (pos >= 0 && pos < count) { 1090 // find next token 1091 pos = readNextToken(srcChars, pos, count, buf, tokenList); 1092 1093 // handle case where end of string is a delimiter 1094 if (pos >= count) { 1095 addToken(tokenList, StringUtils.EMPTY); 1096 } 1097 } 1098 return tokenList; 1099 } 1100 1101 /** 1102 * Gets the String content that the tokenizer is parsing. 1103 * 1104 * @return The string content being parsed. 1105 */ 1106 @Override 1107 public String toString() { 1108 if (tokens == null) { 1109 return "StrTokenizer[not tokenized yet]"; 1110 } 1111 return "StrTokenizer" + getTokenList(); 1112 } 1113 1114}