View Javadoc
1   /*
2    * Licensed to the Apache Software Foundation (ASF) under one or more
3    * contributor license agreements.  See the NOTICE file distributed with
4    * this work for additional information regarding copyright ownership.
5    * The ASF licenses this file to You under the Apache License, Version 2.0
6    * (the "License"); you may not use this file except in compliance with
7    * the License.  You may obtain a copy of the License at
8    *
9    *      https://www.apache.org/licenses/LICENSE-2.0
10   *
11   * Unless required by applicable law or agreed to in writing, software
12   * distributed under the License is distributed on an "AS IS" BASIS,
13   * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
14   * See the License for the specific language governing permissions and
15   * limitations under the License.
16   */
17  package org.apache.commons.lang3.text;
18  
19  import java.util.ArrayList;
20  import java.util.Arrays;
21  import java.util.Collections;
22  import java.util.List;
23  import java.util.ListIterator;
24  import java.util.NoSuchElementException;
25  import java.util.StringTokenizer;
26  
27  import org.apache.commons.lang3.ArrayUtils;
28  import org.apache.commons.lang3.StringUtils;
29  
30  /**
31   * Tokenizes a string based on delimiters (separators)
32   * and supporting quoting and ignored character concepts.
33   * <p>
34   * This class can split a String into many smaller strings. It aims
35   * to do a similar job to {@link java.util.StringTokenizer StringTokenizer},
36   * however it offers much more control and flexibility including implementing
37   * the {@link ListIterator} interface. By default, it is set up
38   * like {@link StringTokenizer}.
39   * </p>
40   * <p>
41   * The input String is split into a number of <em>tokens</em>.
42   * Each token is separated from the next String by a <em>delimiter</em>.
43   * One or more delimiter characters must be specified.
44   * </p>
45   * <p>
46   * Each token may be surrounded by quotes.
47   * The <em>quote</em> matcher specifies the quote character(s).
48   * A quote may be escaped within a quoted section by duplicating itself.
49   * </p>
50   * <p>
51   * Between each token and the delimiter are potentially characters that need trimming.
52   * The <em>trimmer</em> matcher specifies these characters.
53   * One usage might be to trim whitespace characters.
54   * </p>
55   * <p>
56   * At any point outside the quotes there might potentially be invalid characters.
57   * The <em>ignored</em> matcher specifies these characters to be removed.
58   * One usage might be to remove new line characters.
59   * </p>
60   * <p>
61   * Empty tokens may be removed or returned as null.
62   * </p>
63   * <pre>
64   * "a,b,c"         - Three tokens "a","b","c"   (comma delimiter)
65   * " a, b , c "    - Three tokens "a","b","c"   (default CSV processing trims whitespace)
66   * "a, ", b ,", c" - Three tokens "a, " , " b ", ", c" (quoted text untouched)
67   * </pre>
68   *
69   * <table>
70   *  <caption>StrTokenizer properties and options</caption>
71   *  <tr>
72   *   <th>Property</th><th>Type</th><th>Default</th>
73   *  </tr>
74   *  <tr>
75   *   <td>delim</td><td>CharSetMatcher</td><td>{ \t\n\r\f}</td>
76   *  </tr>
77   *  <tr>
78   *   <td>quote</td><td>NoneMatcher</td><td>{}</td>
79   *  </tr>
80   *  <tr>
81   *   <td>ignore</td><td>NoneMatcher</td><td>{}</td>
82   *  </tr>
83   *  <tr>
84   *   <td>emptyTokenAsNull</td><td>boolean</td><td>false</td>
85   *  </tr>
86   *  <tr>
87   *   <td>ignoreEmptyTokens</td><td>boolean</td><td>true</td>
88   *  </tr>
89   * </table>
90   *
91   * @since 2.2
92   * @deprecated As of <a href="https://commons.apache.org/proper/commons-lang/changes-report.html#a3.6">3.6</a>, use Apache Commons Text
93   * <a href="https://commons.apache.org/proper/commons-text/javadocs/api-release/org/apache/commons/text/StringTokenizer.html">
94   * StringTokenizer</a>.
95   */
96  @Deprecated
97  public class StrTokenizer implements ListIterator<String>, Cloneable {
98  
99      // @formatter:off
100     private static final StrTokenizer CSV_TOKENIZER_PROTOTYPE = new StrTokenizer()
101             .setDelimiterMatcher(StrMatcher.commaMatcher())
102             .setQuoteMatcher(StrMatcher.doubleQuoteMatcher())
103             .setIgnoredMatcher(StrMatcher.noneMatcher())
104             .setTrimmerMatcher(StrMatcher.trimMatcher())
105             .setEmptyTokenAsNull(false)
106             .setIgnoreEmptyTokens(false);
107     // @formatter:on
108 
109     // @formatter:off
110     private static final StrTokenizer TSV_TOKENIZER_PROTOTYPE = new StrTokenizer()
111             .setDelimiterMatcher(StrMatcher.tabMatcher())
112             .setQuoteMatcher(StrMatcher.doubleQuoteMatcher())
113             .setIgnoredMatcher(StrMatcher.noneMatcher())
114             .setTrimmerMatcher(StrMatcher.trimMatcher())
115             .setEmptyTokenAsNull(false)
116             .setIgnoreEmptyTokens(false);
117     // @formatter:on
118 
119     /**
120      * Gets a clone of {@code CSV_TOKENIZER_PROTOTYPE}.
121      *
122      * @return A clone of {@code CSV_TOKENIZER_PROTOTYPE}.
123      */
124     private static StrTokenizer getCSVClone() {
125         return (StrTokenizer) CSV_TOKENIZER_PROTOTYPE.clone();
126     }
127 
128     /**
129      * Gets a new tokenizer instance which parses Comma Separated Value strings
130      * initializing it with the given input.  The default for CSV processing
131      * will be trim whitespace from both ends (which can be overridden with
132      * the setTrimmer method).
133      * <p>
134      * You must call a "reset" method to set the string which you want to parse.
135      * </p>
136      *
137      * @return A new tokenizer instance which parses Comma Separated Value strings.
138      */
139     public static StrTokenizer getCSVInstance() {
140         return getCSVClone();
141     }
142 
143     /**
144      * Gets a new tokenizer instance which parses Comma Separated Value strings
145      * initializing it with the given input.  The default for CSV processing
146      * will be trim whitespace from both ends (which can be overridden with
147      * the setTrimmer method).
148      *
149      * @param input  The text to parse.
150      * @return A new tokenizer instance which parses Comma Separated Value strings.
151      */
152     public static StrTokenizer getCSVInstance(final char[] input) {
153         final StrTokenizer tok = getCSVClone();
154         tok.reset(input);
155         return tok;
156     }
157 
158     /**
159      * Gets a new tokenizer instance which parses Comma Separated Value strings
160      * initializing it with the given input.  The default for CSV processing
161      * will be trim whitespace from both ends (which can be overridden with
162      * the setTrimmer method).
163      *
164      * @param input  The text to parse.
165      * @return A new tokenizer instance which parses Comma Separated Value strings.
166      */
167     public static StrTokenizer getCSVInstance(final String input) {
168         final StrTokenizer tok = getCSVClone();
169         tok.reset(input);
170         return tok;
171     }
172 
173     /**
174      * Gets a clone of {@code TSV_TOKENIZER_PROTOTYPE}.
175      *
176      * @return A clone of {@code TSV_TOKENIZER_PROTOTYPE}.
177      */
178     private static StrTokenizer getTSVClone() {
179         return (StrTokenizer) TSV_TOKENIZER_PROTOTYPE.clone();
180     }
181 
182     /**
183      * Gets a new tokenizer instance which parses Tab Separated Value strings.
184      * The default for CSV processing will be trim whitespace from both ends
185      * (which can be overridden with the setTrimmer method).
186      * <p>
187      * You must call a "reset" method to set the string which you want to parse.
188      * </p>
189      *
190      * @return A new tokenizer instance which parses Tab Separated Value strings.
191      */
192     public static StrTokenizer getTSVInstance() {
193         return getTSVClone();
194     }
195 
196     /**
197      * Gets a new tokenizer instance which parses Tab Separated Value strings.
198      * The default for CSV processing will be trim whitespace from both ends
199      * (which can be overridden with the setTrimmer method).
200      *
201      * @param input  The string to parse.
202      * @return A new tokenizer instance which parses Tab Separated Value strings.
203      */
204     public static StrTokenizer getTSVInstance(final char[] input) {
205         final StrTokenizer tok = getTSVClone();
206         tok.reset(input);
207         return tok;
208     }
209 
210     /**
211      * Gets a new tokenizer instance which parses Tab Separated Value strings.
212      * The default for CSV processing will be trim whitespace from both ends
213      * (which can be overridden with the setTrimmer method).
214      *
215      * @param input  The string to parse.
216      * @return A new tokenizer instance which parses Tab Separated Value strings.
217      */
218     public static StrTokenizer getTSVInstance(final String input) {
219         final StrTokenizer tok = getTSVClone();
220         tok.reset(input);
221         return tok;
222     }
223 
224     /** The text to work on. */
225     private char[] chars;
226 
227     /** The parsed tokens */
228     private String[] tokens;
229 
230     /** The current iteration position */
231     private int tokenPos;
232 
233     /** The delimiter matcher */
234     private StrMatcher delimMatcher = StrMatcher.splitMatcher();
235 
236     /** The quote matcher */
237     private StrMatcher quoteMatcher = StrMatcher.noneMatcher();
238 
239     /** The ignored matcher */
240     private StrMatcher ignoredMatcher = StrMatcher.noneMatcher();
241 
242     /** The trimmer matcher */
243     private StrMatcher trimmerMatcher = StrMatcher.noneMatcher();
244 
245     /** Whether to return empty tokens as null */
246     private boolean emptyAsNull;
247 
248     /** Whether to ignore empty tokens */
249     private boolean ignoreEmptyTokens = true;
250 
251     /**
252      * Constructs a tokenizer splitting on space, tab, newline and formfeed
253      * as per StringTokenizer, but with no text to tokenize.
254      * <p>
255      * This constructor is normally used with {@link #reset(String)}.
256      * </p>
257      */
258     public StrTokenizer() {
259         this.chars = null;
260     }
261 
262     /**
263      * Constructs a tokenizer splitting on space, tab, newline and formfeed
264      * as per StringTokenizer.
265      *
266      * @param input  The string which is to be parsed, not cloned.
267      */
268     public StrTokenizer(final char[] input) {
269         this.chars = ArrayUtils.clone(input);
270     }
271 
272     /**
273      * Constructs a tokenizer splitting on the specified character.
274      *
275      * @param input  The string which is to be parsed, not cloned.
276      * @param delim The field delimiter character.
277      */
278     public StrTokenizer(final char[] input, final char delim) {
279         this(input);
280         setDelimiterChar(delim);
281     }
282 
283     /**
284      * Constructs a tokenizer splitting on the specified delimiter character
285      * and handling quotes using the specified quote character.
286      *
287      * @param input  The string which is to be parsed, not cloned.
288      * @param delim  The field delimiter character.
289      * @param quote  The field quoted string character.
290      */
291     public StrTokenizer(final char[] input, final char delim, final char quote) {
292         this(input, delim);
293         setQuoteChar(quote);
294     }
295 
296     /**
297      * Constructs a tokenizer splitting on the specified string.
298      *
299      * @param input  The string which is to be parsed, not cloned.
300      * @param delim The field delimiter string.
301      */
302     public StrTokenizer(final char[] input, final String delim) {
303         this(input);
304         setDelimiterString(delim);
305     }
306 
307     /**
308      * Constructs a tokenizer splitting using the specified delimiter matcher.
309      *
310      * @param input  The string which is to be parsed, not cloned.
311      * @param delim  The field delimiter matcher.
312      */
313     public StrTokenizer(final char[] input, final StrMatcher delim) {
314         this(input);
315         setDelimiterMatcher(delim);
316     }
317 
318     /**
319      * Constructs a tokenizer splitting using the specified delimiter matcher
320      * and handling quotes using the specified quote matcher.
321      *
322      * @param input  The string which is to be parsed, not cloned.
323      * @param delim  The field delimiter character.
324      * @param quote  The field quoted string character.
325      */
326     public StrTokenizer(final char[] input, final StrMatcher delim, final StrMatcher quote) {
327         this(input, delim);
328         setQuoteMatcher(quote);
329     }
330 
331     /**
332      * Constructs a tokenizer splitting on space, tab, newline and formfeed
333      * as per StringTokenizer.
334      *
335      * @param input  The string which is to be parsed.
336      */
337     public StrTokenizer(final String input) {
338         if (input != null) {
339             chars = input.toCharArray();
340         } else {
341             chars = null;
342         }
343     }
344 
345     /**
346      * Constructs a tokenizer splitting on the specified delimiter character.
347      *
348      * @param input  The string which is to be parsed.
349      * @param delim  The field delimiter character.
350      */
351     public StrTokenizer(final String input, final char delim) {
352         this(input);
353         setDelimiterChar(delim);
354     }
355 
356     /**
357      * Constructs a tokenizer splitting on the specified delimiter character
358      * and handling quotes using the specified quote character.
359      *
360      * @param input  The string which is to be parsed.
361      * @param delim  The field delimiter character.
362      * @param quote  The field quoted string character.
363      */
364     public StrTokenizer(final String input, final char delim, final char quote) {
365         this(input, delim);
366         setQuoteChar(quote);
367     }
368 
369     /**
370      * Constructs a tokenizer splitting on the specified delimiter string.
371      *
372      * @param input  The string which is to be parsed.
373      * @param delim  The field delimiter string.
374      */
375     public StrTokenizer(final String input, final String delim) {
376         this(input);
377         setDelimiterString(delim);
378     }
379 
380     /**
381      * Constructs a tokenizer splitting using the specified delimiter matcher.
382      *
383      * @param input  The string which is to be parsed.
384      * @param delim  The field delimiter matcher.
385      */
386     public StrTokenizer(final String input, final StrMatcher delim) {
387         this(input);
388         setDelimiterMatcher(delim);
389     }
390 
391     /**
392      * Constructs a tokenizer splitting using the specified delimiter matcher
393      * and handling quotes using the specified quote matcher.
394      *
395      * @param input  The string which is to be parsed.
396      * @param delim  The field delimiter matcher.
397      * @param quote  The field quoted string matcher.
398      */
399     public StrTokenizer(final String input, final StrMatcher delim, final StrMatcher quote) {
400         this(input, delim);
401         setQuoteMatcher(quote);
402     }
403 
404     /**
405      * Always throws {@link UnsupportedOperationException}.
406      *
407      * @param obj this parameter ignored.
408      * @throws UnsupportedOperationException Thrown because this operation is unsupported.
409      */
410     @Override
411     public void add(final String obj) {
412         throw new UnsupportedOperationException("add() is unsupported");
413     }
414 
415     /**
416      * Adds a token to a list, paying attention to the parameters we've set.
417      *
418      * @param list  The list to add to.
419      * @param tok  The token to add.
420      */
421     private void addToken(final List<String> list, String tok) {
422         if (StringUtils.isEmpty(tok)) {
423             if (isIgnoreEmptyTokens()) {
424                 return;
425             }
426             if (isEmptyTokenAsNull()) {
427                 tok = null;
428             }
429         }
430         list.add(tok);
431     }
432 
433     /**
434      * Checks if tokenization has been done, and if not then do it.
435      */
436     private void checkTokenized() {
437         if (tokens == null) {
438             if (chars == null) {
439                 // still call tokenize as subclass may do some work
440                 final List<String> split = tokenize(null, 0, 0);
441                 tokens = split.toArray(ArrayUtils.EMPTY_STRING_ARRAY);
442             } else {
443                 final List<String> split = tokenize(chars, 0, chars.length);
444                 tokens = split.toArray(ArrayUtils.EMPTY_STRING_ARRAY);
445             }
446         }
447     }
448 
449     /**
450      * Creates a new instance of this Tokenizer. The new instance is reset so
451      * that it will be at the start of the token list.
452      * If a {@link CloneNotSupportedException} is caught, return {@code null}.
453      *
454      * @return A new instance of this Tokenizer which has been reset.
455      */
456     @Override
457     public Object clone() {
458         try {
459             return cloneReset();
460         } catch (final CloneNotSupportedException ex) {
461             return null;
462         }
463     }
464 
465     /**
466      * Creates a new instance of this Tokenizer. The new instance is reset so that
467      * it will be at the start of the token list.
468      *
469      * @return A new instance of this Tokenizer which has been reset.
470      * @throws CloneNotSupportedException Thrown if there is a problem cloning.
471      */
472     Object cloneReset() throws CloneNotSupportedException {
473         // this method exists to enable 100% test coverage
474         final StrTokenizer cloned = (StrTokenizer) super.clone();
475         if (cloned.chars != null) {
476             cloned.chars = cloned.chars.clone();
477         }
478         cloned.reset();
479         return cloned;
480     }
481 
482     /**
483      * Gets the String content that the tokenizer is parsing.
484      *
485      * @return The string content being parsed.
486      */
487     public String getContent() {
488         if (chars == null) {
489             return null;
490         }
491         return new String(chars);
492     }
493 
494     /**
495      * Gets the field delimiter matcher.
496      *
497      * @return The delimiter matcher in use.
498      */
499     public StrMatcher getDelimiterMatcher() {
500         return this.delimMatcher;
501     }
502 
503     /**
504      * Gets the ignored character matcher.
505      * <p>
506      * These characters are ignored when parsing the String, unless they are
507      * within a quoted region.
508      * The default value is not to ignore anything.
509      * </p>
510      *
511      * @return The ignored matcher in use.
512      */
513     public StrMatcher getIgnoredMatcher() {
514         return ignoredMatcher;
515     }
516 
517     /**
518      * Gets the quote matcher currently in use.
519      * <p>
520      * The quote character is used to wrap data between the tokens.
521      * This enables delimiters to be entered as data.
522      * The default value is '"' (double quote).
523      * </p>
524      *
525      * @return The quote matcher in use.
526      */
527     public StrMatcher getQuoteMatcher() {
528         return quoteMatcher;
529     }
530 
531     /**
532      * Gets a copy of the full token list as an independent modifiable array.
533      *
534      * @return The tokens as a String array.
535      */
536     public String[] getTokenArray() {
537         checkTokenized();
538         return tokens.clone();
539     }
540 
541     /**
542      * Gets a copy of the full token list as an independent modifiable list.
543      *
544      * @return The tokens as a String array.
545      */
546     public List<String> getTokenList() {
547         checkTokenized();
548         final List<String> list = new ArrayList<>(tokens.length);
549         list.addAll(Arrays.asList(tokens));
550         return list;
551     }
552 
553     /**
554      * Gets the trimmer character matcher.
555      * <p>
556      * These characters are trimmed off on each side of the delimiter
557      * until the token or quote is found.
558      * The default value is not to trim anything.
559      * </p>
560      *
561      * @return The trimmer matcher in use.
562      */
563     public StrMatcher getTrimmerMatcher() {
564         return trimmerMatcher;
565     }
566 
567     /**
568      * Tests whether there are any more tokens.
569      *
570      * @return true if there are more tokens.
571      */
572     @Override
573     public boolean hasNext() {
574         checkTokenized();
575         return tokenPos < tokens.length;
576     }
577 
578     /**
579      * Tests whether there are any previous tokens that can be iterated to.
580      *
581      * @return true if there are previous tokens.
582      */
583     @Override
584     public boolean hasPrevious() {
585         checkTokenized();
586         return tokenPos > 0;
587     }
588 
589     /**
590      * Tests whether the tokenizer currently returns empty tokens as null. The default for this property is false.
591      *
592      * @return true if empty tokens are returned as null.
593      */
594     public boolean isEmptyTokenAsNull() {
595         return this.emptyAsNull;
596     }
597 
598     /**
599      * Tests whether the tokenizer currently ignores empty tokens. The default for this property is true.
600      *
601      * @return true if empty tokens are not returned.
602      */
603     public boolean isIgnoreEmptyTokens() {
604         return ignoreEmptyTokens;
605     }
606 
607     /**
608      * Tests whether the characters at the index specified match the quote already matched in readNextToken().
609      *
610      * @param srcChars  The character array being tokenized.
611      * @param pos  The position to check for a quote.
612      * @param len  The length of the character array being tokenized.
613      * @param quoteStart  The start position of the matched quote, 0 if no quoting.
614      * @param quoteLen  The length of the matched quote, 0 if no quoting.
615      * @return true if a quote is matched.
616      */
617     private boolean isQuote(final char[] srcChars, final int pos, final int len, final int quoteStart, final int quoteLen) {
618         for (int i = 0; i < quoteLen; i++) {
619             if (pos + i >= len || srcChars[pos + i] != srcChars[quoteStart + i]) {
620                 return false;
621             }
622         }
623         return true;
624     }
625 
626     /**
627      * Gets the next token.
628      *
629      * @return The next String token.
630      * @throws NoSuchElementException Thrown if there are no more elements.
631      */
632     @Override
633     public String next() {
634         if (hasNext()) {
635             return tokens[tokenPos++];
636         }
637         throw new NoSuchElementException();
638     }
639 
640     /**
641      * Gets the index of the next token to return.
642      *
643      * @return The next token index.
644      */
645     @Override
646     public int nextIndex() {
647         return tokenPos;
648     }
649 
650     /**
651      * Gets the next token from the String.
652      * Equivalent to {@link #next()} except it returns null rather than
653      * throwing {@link NoSuchElementException} when no tokens remain.
654      *
655      * @return The next sequential token, or null when no more tokens are found.
656      */
657     public String nextToken() {
658         if (hasNext()) {
659             return tokens[tokenPos++];
660         }
661         return null;
662     }
663 
664     /**
665      * Gets the token previous to the last returned token.
666      *
667      * @return The previous token.
668      */
669     @Override
670     public String previous() {
671         if (hasPrevious()) {
672             return tokens[--tokenPos];
673         }
674         throw new NoSuchElementException();
675     }
676 
677     /**
678      * Gets the index of the previous token.
679      *
680      * @return The previous token index.
681      */
682     @Override
683     public int previousIndex() {
684         return tokenPos - 1;
685     }
686 
687     /**
688      * Gets the previous token from the String.
689      *
690      * @return The previous sequential token, or null when no more tokens are found.
691      */
692     public String previousToken() {
693         if (hasPrevious()) {
694             return tokens[--tokenPos];
695         }
696         return null;
697     }
698 
699     /**
700      * Reads character by character through the String to get the next token.
701      *
702      * @param srcChars  The character array being tokenized.
703      * @param start  The first character of field.
704      * @param len  The length of the character array being tokenized.
705      * @param workArea  A temporary work area.
706      * @param tokenList  The list of parsed tokens.
707      * @return The starting position of the next field (the character
708      *  immediately after the delimiter), or -1 if end of string found.
709      */
710     private int readNextToken(final char[] srcChars, int start, final int len, final StrBuilder workArea, final List<String> tokenList) {
711         // skip all leading whitespace, unless it is the
712         // field delimiter or the quote character
713         while (start < len) {
714             final int removeLen = Math.max(
715                     getIgnoredMatcher().isMatch(srcChars, start, start, len),
716                     getTrimmerMatcher().isMatch(srcChars, start, start, len));
717             if (removeLen == 0 ||
718                 getDelimiterMatcher().isMatch(srcChars, start, start, len) > 0 ||
719                 getQuoteMatcher().isMatch(srcChars, start, start, len) > 0) {
720                 break;
721             }
722             start += removeLen;
723         }
724 
725         // handle reaching end
726         if (start >= len) {
727             addToken(tokenList, StringUtils.EMPTY);
728             return -1;
729         }
730 
731         // handle empty token
732         final int delimLen = getDelimiterMatcher().isMatch(srcChars, start, start, len);
733         if (delimLen > 0) {
734             addToken(tokenList, StringUtils.EMPTY);
735             return start + delimLen;
736         }
737 
738         // handle found token
739         final int quoteLen = getQuoteMatcher().isMatch(srcChars, start, start, len);
740         if (quoteLen > 0) {
741             return readWithQuotes(srcChars, start + quoteLen, len, workArea, tokenList, start, quoteLen);
742         }
743         return readWithQuotes(srcChars, start, len, workArea, tokenList, 0, 0);
744     }
745 
746     /**
747      * Reads a possibly quoted string token.
748      *
749      * @param srcChars  The character array being tokenized.
750      * @param start  The first character of field.
751      * @param len  The length of the character array being tokenized.
752      * @param workArea  A temporary work area.
753      * @param tokenList  The list of parsed tokens.
754      * @param quoteStart  The start position of the matched quote, 0 if no quoting.
755      * @param quoteLen  The length of the matched quote, 0 if no quoting.
756      * @return The starting position of the next field (the character
757      *  immediately after the delimiter, or if end of string found,
758      *  then the length of string.
759      */
760     private int readWithQuotes(final char[] srcChars, final int start, final int len, final StrBuilder workArea,
761                                final List<String> tokenList, final int quoteStart, final int quoteLen) {
762         // Loop until we've found the end of the quoted
763         // string or the end of the input
764         workArea.clear();
765         int pos = start;
766         boolean quoting = quoteLen > 0;
767         int trimStart = 0;
768 
769         while (pos < len) {
770             // quoting mode can occur several times throughout a string
771             // we must switch between quoting and non-quoting until we
772             // encounter a non-quoted delimiter, or end of string
773             if (quoting) {
774                 // In quoting mode
775 
776                 // If we've found a quote character, see if it's
777                 // followed by a second quote.  If so, then we need
778                 // to actually put the quote character into the token
779                 // rather than end the token.
780                 if (isQuote(srcChars, pos, len, quoteStart, quoteLen)) {
781                     if (isQuote(srcChars, pos + quoteLen, len, quoteStart, quoteLen)) {
782                         // matched pair of quotes, thus an escaped quote
783                         workArea.append(srcChars, pos, quoteLen);
784                         pos += quoteLen * 2;
785                         trimStart = workArea.size();
786                         continue;
787                     }
788 
789                     // end of quoting
790                     quoting = false;
791                     pos += quoteLen;
792                     continue;
793                 }
794 
795             } else {
796                 // Not in quoting mode
797 
798                 // check for delimiter, and thus end of token
799                 final int delimLen = getDelimiterMatcher().isMatch(srcChars, pos, start, len);
800                 if (delimLen > 0) {
801                     // return condition when end of token found
802                     addToken(tokenList, workArea.substring(0, trimStart));
803                     return pos + delimLen;
804                 }
805 
806                 // check for quote, and thus back into quoting mode
807                 if (quoteLen > 0 && isQuote(srcChars, pos, len, quoteStart, quoteLen)) {
808                     quoting = true;
809                     pos += quoteLen;
810                     continue;
811                 }
812 
813                 // check for ignored (outside quotes), and ignore
814                 final int ignoredLen = getIgnoredMatcher().isMatch(srcChars, pos, start, len);
815                 if (ignoredLen > 0) {
816                     pos += ignoredLen;
817                     continue;
818                 }
819 
820                 // check for trimmed character
821                 // don't yet know if it's at the end, so copy to workArea
822                 // use trimStart to keep track of trim at the end
823                 final int trimmedLen = getTrimmerMatcher().isMatch(srcChars, pos, start, len);
824                 if (trimmedLen > 0) {
825                     workArea.append(srcChars, pos, trimmedLen);
826                     pos += trimmedLen;
827                     continue;
828                 }
829             }
830             // copy regular character from inside quotes
831             workArea.append(srcChars[pos++]);
832             trimStart = workArea.size();
833         }
834 
835         // return condition when end of string found
836         addToken(tokenList, workArea.substring(0, trimStart));
837         return -1;
838     }
839 
840     /**
841      * Always throws {@link UnsupportedOperationException}.
842      *
843      * @throws UnsupportedOperationException Thrown because this operation is unsupported.
844      */
845     @Override
846     public void remove() {
847         throw new UnsupportedOperationException("remove() is unsupported");
848     }
849 
850     /**
851      * Resets this tokenizer, forgetting all parsing and iteration already completed.
852      * <p>
853      * This method allows the same tokenizer to be reused for the same String.
854      * </p>
855      *
856      * @return {@code this} instance.
857      */
858     public StrTokenizer reset() {
859         tokenPos = 0;
860         tokens = null;
861         return this;
862     }
863 
864     /**
865      * Reset this tokenizer, giving it a new input string to parse.
866      * In this manner you can re-use a tokenizer with the same settings
867      * on multiple input lines.
868      *
869      * @param input  The new character array to tokenize, not cloned, null sets no text to parse.
870      * @return {@code this} instance.
871      */
872     public StrTokenizer reset(final char[] input) {
873         reset();
874         this.chars = ArrayUtils.clone(input);
875         return this;
876     }
877 
878     /**
879      * Reset this tokenizer, giving it a new input string to parse.
880      * In this manner you can re-use a tokenizer with the same settings
881      * on multiple input lines.
882      *
883      * @param input  The new string to tokenize, null sets no text to parse.
884      * @return {@code this} instance.
885      */
886     public StrTokenizer reset(final String input) {
887         reset();
888         if (input != null) {
889             this.chars = input.toCharArray();
890         } else {
891             this.chars = null;
892         }
893         return this;
894     }
895 
896     /**
897      * Sets no token and always throws {@link UnsupportedOperationException}.
898      *
899      * @param obj this parameter ignored.
900      * @throws UnsupportedOperationException Thrown because this operation is unsupported.
901      */
902     @Override
903     public void set(final String obj) {
904         throw new UnsupportedOperationException("set() is unsupported");
905     }
906 
907     /**
908      * Sets the field delimiter character.
909      *
910      * @param delim  The delimiter character to use.
911      * @return {@code this} instance.
912      */
913     public StrTokenizer setDelimiterChar(final char delim) {
914         return setDelimiterMatcher(StrMatcher.charMatcher(delim));
915     }
916 
917     /**
918      * Sets the field delimiter matcher.
919      * <p>
920      * The delimiter is used to separate one token from another.
921      * </p>
922      *
923      * @param delim  The delimiter matcher to use.
924      * @return {@code this} instance.
925      */
926     public StrTokenizer setDelimiterMatcher(final StrMatcher delim) {
927         if (delim == null) {
928             this.delimMatcher = StrMatcher.noneMatcher();
929         } else {
930             this.delimMatcher = delim;
931         }
932         return this;
933     }
934 
935     /**
936      * Sets the field delimiter string.
937      *
938      * @param delim  The delimiter string to use.
939      * @return {@code this} instance.
940      */
941     public StrTokenizer setDelimiterString(final String delim) {
942         return setDelimiterMatcher(StrMatcher.stringMatcher(delim));
943     }
944 
945     /**
946      * Sets whether the tokenizer should return empty tokens as null.
947      * The default for this property is false.
948      *
949      * @param emptyAsNull  whether empty tokens are returned as null.
950      * @return {@code this} instance.
951      */
952     public StrTokenizer setEmptyTokenAsNull(final boolean emptyAsNull) {
953         this.emptyAsNull = emptyAsNull;
954         return this;
955     }
956 
957     /**
958      * Sets the character to ignore.
959      * <p>
960      * This character is ignored when parsing the String, unless it is
961      * within a quoted region.
962      *
963      * @param ignored  The ignored character to use.
964      * @return {@code this} instance.
965      */
966     public StrTokenizer setIgnoredChar(final char ignored) {
967         return setIgnoredMatcher(StrMatcher.charMatcher(ignored));
968     }
969 
970     /**
971      * Sets the matcher for characters to ignore.
972      * <p>
973      * These characters are ignored when parsing the String, unless they are
974      * within a quoted region.
975      * </p>
976      *
977      * @param ignored  The ignored matcher to use, null ignored.
978      * @return {@code this} instance.
979      */
980     public StrTokenizer setIgnoredMatcher(final StrMatcher ignored) {
981         if (ignored != null) {
982             this.ignoredMatcher = ignored;
983         }
984         return this;
985     }
986 
987     /**
988      * Sets whether the tokenizer should ignore and not return empty tokens.
989      * The default for this property is true.
990      *
991      * @param ignoreEmptyTokens  whether empty tokens are not returned.
992      * @return {@code this} instance.
993      */
994     public StrTokenizer setIgnoreEmptyTokens(final boolean ignoreEmptyTokens) {
995         this.ignoreEmptyTokens = ignoreEmptyTokens;
996         return this;
997     }
998 
999     /**
1000      * Sets the quote character to use.
1001      * <p>
1002      * The quote character is used to wrap data between the tokens.
1003      * This enables delimiters to be entered as data.
1004      * </p>
1005      *
1006      * @param quote  The quote character to use.
1007      * @return {@code this} instance.
1008      */
1009     public StrTokenizer setQuoteChar(final char quote) {
1010         return setQuoteMatcher(StrMatcher.charMatcher(quote));
1011     }
1012 
1013     /**
1014      * Sets the quote matcher to use.
1015      * <p>
1016      * The quote character is used to wrap data between the tokens.
1017      * This enables delimiters to be entered as data.
1018      * </p>
1019      *
1020      * @param quote  The quote matcher to use, null ignored.
1021      * @return {@code this} instance.
1022      */
1023     public StrTokenizer setQuoteMatcher(final StrMatcher quote) {
1024         if (quote != null) {
1025             this.quoteMatcher = quote;
1026         }
1027         return this;
1028     }
1029 
1030     /**
1031      * Sets the matcher for characters to trim.
1032      * <p>
1033      * These characters are trimmed off on each side of the delimiter
1034      * until the token or quote is found.
1035      * </p>
1036      *
1037      * @param trimmer  The trimmer matcher to use, null ignored.
1038      * @return {@code this} instance.
1039      */
1040     public StrTokenizer setTrimmerMatcher(final StrMatcher trimmer) {
1041         if (trimmer != null) {
1042             this.trimmerMatcher = trimmer;
1043         }
1044         return this;
1045     }
1046 
1047     /**
1048      * Gets the number of tokens found in the String.
1049      *
1050      * @return The number of matched tokens.
1051      */
1052     public int size() {
1053         checkTokenized();
1054         return tokens.length;
1055     }
1056 
1057     /**
1058      * Internal method to performs the tokenization.
1059      * <p>
1060      * Most users of this class do not need to call this method. This method
1061      * will be called automatically by other (public) methods when required.
1062      * </p>
1063      * <p>
1064      * This method exists to allow subclasses to add code before or after the
1065      * tokenization. For example, a subclass could alter the character array,
1066      * offset or count to be parsed, or call the tokenizer multiple times on
1067      * multiple strings. It is also be possible to filter the results.
1068      * </p>
1069      * <p>
1070      * {@link StrTokenizer} will always pass a zero offset and a count
1071      * equal to the length of the array to this method, however a subclass
1072      * may pass other values, or even an entirely different array.
1073      * </p>
1074      *
1075      * @param srcChars  The character array being tokenized, may be null.
1076      * @param offset  The start position within the character array, must be valid.
1077      * @param count  The number of characters to tokenize, must be valid.
1078      * @return The modifiable list of String tokens, unmodifiable if null array or zero count.
1079      */
1080     protected List<String> tokenize(final char[] srcChars, final int offset, final int count) {
1081         if (ArrayUtils.isEmpty(srcChars)) {
1082             return Collections.emptyList();
1083         }
1084         final StrBuilder buf = new StrBuilder();
1085         final List<String> tokenList = new ArrayList<>();
1086         int pos = offset;
1087 
1088         // loop around the entire buffer
1089         while (pos >= 0 && pos < count) {
1090             // find next token
1091             pos = readNextToken(srcChars, pos, count, buf, tokenList);
1092 
1093             // handle case where end of string is a delimiter
1094             if (pos >= count) {
1095                 addToken(tokenList, StringUtils.EMPTY);
1096             }
1097         }
1098         return tokenList;
1099     }
1100 
1101     /**
1102      * Gets the String content that the tokenizer is parsing.
1103      *
1104      * @return The string content being parsed.
1105      */
1106     @Override
1107     public String toString() {
1108         if (tokens == null) {
1109             return "StrTokenizer[not tokenized yet]";
1110         }
1111         return "StrTokenizer" + getTokenList();
1112     }
1113 
1114 }