1 /*
2 * Licensed to the Apache Software Foundation (ASF) under one or more
3 * contributor license agreements. See the NOTICE file distributed with
4 * this work for additional information regarding copyright ownership.
5 * The ASF licenses this file to You under the Apache License, Version 2.0
6 * (the "License"); you may not use this file except in compliance with
7 * the License. You may obtain a copy of the License at
8 *
9 * https://www.apache.org/licenses/LICENSE-2.0
10 *
11 * Unless required by applicable law or agreed to in writing, software
12 * distributed under the License is distributed on an "AS IS" BASIS,
13 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
14 * See the License for the specific language governing permissions and
15 * limitations under the License.
16 */
17 package org.apache.commons.lang3.text;
18
19 import java.util.ArrayList;
20 import java.util.Arrays;
21 import java.util.Collections;
22 import java.util.List;
23 import java.util.ListIterator;
24 import java.util.NoSuchElementException;
25 import java.util.StringTokenizer;
26
27 import org.apache.commons.lang3.ArrayUtils;
28 import org.apache.commons.lang3.StringUtils;
29
30 /**
31 * Tokenizes a string based on delimiters (separators)
32 * and supporting quoting and ignored character concepts.
33 * <p>
34 * This class can split a String into many smaller strings. It aims
35 * to do a similar job to {@link java.util.StringTokenizer StringTokenizer},
36 * however it offers much more control and flexibility including implementing
37 * the {@link ListIterator} interface. By default, it is set up
38 * like {@link StringTokenizer}.
39 * </p>
40 * <p>
41 * The input String is split into a number of <em>tokens</em>.
42 * Each token is separated from the next String by a <em>delimiter</em>.
43 * One or more delimiter characters must be specified.
44 * </p>
45 * <p>
46 * Each token may be surrounded by quotes.
47 * The <em>quote</em> matcher specifies the quote character(s).
48 * A quote may be escaped within a quoted section by duplicating itself.
49 * </p>
50 * <p>
51 * Between each token and the delimiter are potentially characters that need trimming.
52 * The <em>trimmer</em> matcher specifies these characters.
53 * One usage might be to trim whitespace characters.
54 * </p>
55 * <p>
56 * At any point outside the quotes there might potentially be invalid characters.
57 * The <em>ignored</em> matcher specifies these characters to be removed.
58 * One usage might be to remove new line characters.
59 * </p>
60 * <p>
61 * Empty tokens may be removed or returned as null.
62 * </p>
63 * <pre>
64 * "a,b,c" - Three tokens "a","b","c" (comma delimiter)
65 * " a, b , c " - Three tokens "a","b","c" (default CSV processing trims whitespace)
66 * "a, ", b ,", c" - Three tokens "a, " , " b ", ", c" (quoted text untouched)
67 * </pre>
68 *
69 * <table>
70 * <caption>StrTokenizer properties and options</caption>
71 * <tr>
72 * <th>Property</th><th>Type</th><th>Default</th>
73 * </tr>
74 * <tr>
75 * <td>delim</td><td>CharSetMatcher</td><td>{ \t\n\r\f}</td>
76 * </tr>
77 * <tr>
78 * <td>quote</td><td>NoneMatcher</td><td>{}</td>
79 * </tr>
80 * <tr>
81 * <td>ignore</td><td>NoneMatcher</td><td>{}</td>
82 * </tr>
83 * <tr>
84 * <td>emptyTokenAsNull</td><td>boolean</td><td>false</td>
85 * </tr>
86 * <tr>
87 * <td>ignoreEmptyTokens</td><td>boolean</td><td>true</td>
88 * </tr>
89 * </table>
90 *
91 * @since 2.2
92 * @deprecated As of <a href="https://commons.apache.org/proper/commons-lang/changes-report.html#a3.6">3.6</a>, use Apache Commons Text
93 * <a href="https://commons.apache.org/proper/commons-text/javadocs/api-release/org/apache/commons/text/StringTokenizer.html">
94 * StringTokenizer</a>.
95 */
96 @Deprecated
97 public class StrTokenizer implements ListIterator<String>, Cloneable {
98
99 // @formatter:off
100 private static final StrTokenizer CSV_TOKENIZER_PROTOTYPE = new StrTokenizer()
101 .setDelimiterMatcher(StrMatcher.commaMatcher())
102 .setQuoteMatcher(StrMatcher.doubleQuoteMatcher())
103 .setIgnoredMatcher(StrMatcher.noneMatcher())
104 .setTrimmerMatcher(StrMatcher.trimMatcher())
105 .setEmptyTokenAsNull(false)
106 .setIgnoreEmptyTokens(false);
107 // @formatter:on
108
109 // @formatter:off
110 private static final StrTokenizer TSV_TOKENIZER_PROTOTYPE = new StrTokenizer()
111 .setDelimiterMatcher(StrMatcher.tabMatcher())
112 .setQuoteMatcher(StrMatcher.doubleQuoteMatcher())
113 .setIgnoredMatcher(StrMatcher.noneMatcher())
114 .setTrimmerMatcher(StrMatcher.trimMatcher())
115 .setEmptyTokenAsNull(false)
116 .setIgnoreEmptyTokens(false);
117 // @formatter:on
118
119 /**
120 * Gets a clone of {@code CSV_TOKENIZER_PROTOTYPE}.
121 *
122 * @return A clone of {@code CSV_TOKENIZER_PROTOTYPE}.
123 */
124 private static StrTokenizer getCSVClone() {
125 return (StrTokenizer) CSV_TOKENIZER_PROTOTYPE.clone();
126 }
127
128 /**
129 * Gets a new tokenizer instance which parses Comma Separated Value strings
130 * initializing it with the given input. The default for CSV processing
131 * will be trim whitespace from both ends (which can be overridden with
132 * the setTrimmer method).
133 * <p>
134 * You must call a "reset" method to set the string which you want to parse.
135 * </p>
136 *
137 * @return A new tokenizer instance which parses Comma Separated Value strings.
138 */
139 public static StrTokenizer getCSVInstance() {
140 return getCSVClone();
141 }
142
143 /**
144 * Gets a new tokenizer instance which parses Comma Separated Value strings
145 * initializing it with the given input. The default for CSV processing
146 * will be trim whitespace from both ends (which can be overridden with
147 * the setTrimmer method).
148 *
149 * @param input The text to parse.
150 * @return A new tokenizer instance which parses Comma Separated Value strings.
151 */
152 public static StrTokenizer getCSVInstance(final char[] input) {
153 final StrTokenizer tok = getCSVClone();
154 tok.reset(input);
155 return tok;
156 }
157
158 /**
159 * Gets a new tokenizer instance which parses Comma Separated Value strings
160 * initializing it with the given input. The default for CSV processing
161 * will be trim whitespace from both ends (which can be overridden with
162 * the setTrimmer method).
163 *
164 * @param input The text to parse.
165 * @return A new tokenizer instance which parses Comma Separated Value strings.
166 */
167 public static StrTokenizer getCSVInstance(final String input) {
168 final StrTokenizer tok = getCSVClone();
169 tok.reset(input);
170 return tok;
171 }
172
173 /**
174 * Gets a clone of {@code TSV_TOKENIZER_PROTOTYPE}.
175 *
176 * @return A clone of {@code TSV_TOKENIZER_PROTOTYPE}.
177 */
178 private static StrTokenizer getTSVClone() {
179 return (StrTokenizer) TSV_TOKENIZER_PROTOTYPE.clone();
180 }
181
182 /**
183 * Gets a new tokenizer instance which parses Tab Separated Value strings.
184 * The default for CSV processing will be trim whitespace from both ends
185 * (which can be overridden with the setTrimmer method).
186 * <p>
187 * You must call a "reset" method to set the string which you want to parse.
188 * </p>
189 *
190 * @return A new tokenizer instance which parses Tab Separated Value strings.
191 */
192 public static StrTokenizer getTSVInstance() {
193 return getTSVClone();
194 }
195
196 /**
197 * Gets a new tokenizer instance which parses Tab Separated Value strings.
198 * The default for CSV processing will be trim whitespace from both ends
199 * (which can be overridden with the setTrimmer method).
200 *
201 * @param input The string to parse.
202 * @return A new tokenizer instance which parses Tab Separated Value strings.
203 */
204 public static StrTokenizer getTSVInstance(final char[] input) {
205 final StrTokenizer tok = getTSVClone();
206 tok.reset(input);
207 return tok;
208 }
209
210 /**
211 * Gets a new tokenizer instance which parses Tab Separated Value strings.
212 * The default for CSV processing will be trim whitespace from both ends
213 * (which can be overridden with the setTrimmer method).
214 *
215 * @param input The string to parse.
216 * @return A new tokenizer instance which parses Tab Separated Value strings.
217 */
218 public static StrTokenizer getTSVInstance(final String input) {
219 final StrTokenizer tok = getTSVClone();
220 tok.reset(input);
221 return tok;
222 }
223
224 /** The text to work on. */
225 private char[] chars;
226
227 /** The parsed tokens */
228 private String[] tokens;
229
230 /** The current iteration position */
231 private int tokenPos;
232
233 /** The delimiter matcher */
234 private StrMatcher delimMatcher = StrMatcher.splitMatcher();
235
236 /** The quote matcher */
237 private StrMatcher quoteMatcher = StrMatcher.noneMatcher();
238
239 /** The ignored matcher */
240 private StrMatcher ignoredMatcher = StrMatcher.noneMatcher();
241
242 /** The trimmer matcher */
243 private StrMatcher trimmerMatcher = StrMatcher.noneMatcher();
244
245 /** Whether to return empty tokens as null */
246 private boolean emptyAsNull;
247
248 /** Whether to ignore empty tokens */
249 private boolean ignoreEmptyTokens = true;
250
251 /**
252 * Constructs a tokenizer splitting on space, tab, newline and formfeed
253 * as per StringTokenizer, but with no text to tokenize.
254 * <p>
255 * This constructor is normally used with {@link #reset(String)}.
256 * </p>
257 */
258 public StrTokenizer() {
259 this.chars = null;
260 }
261
262 /**
263 * Constructs a tokenizer splitting on space, tab, newline and formfeed
264 * as per StringTokenizer.
265 *
266 * @param input The string which is to be parsed, not cloned.
267 */
268 public StrTokenizer(final char[] input) {
269 this.chars = ArrayUtils.clone(input);
270 }
271
272 /**
273 * Constructs a tokenizer splitting on the specified character.
274 *
275 * @param input The string which is to be parsed, not cloned.
276 * @param delim The field delimiter character.
277 */
278 public StrTokenizer(final char[] input, final char delim) {
279 this(input);
280 setDelimiterChar(delim);
281 }
282
283 /**
284 * Constructs a tokenizer splitting on the specified delimiter character
285 * and handling quotes using the specified quote character.
286 *
287 * @param input The string which is to be parsed, not cloned.
288 * @param delim The field delimiter character.
289 * @param quote The field quoted string character.
290 */
291 public StrTokenizer(final char[] input, final char delim, final char quote) {
292 this(input, delim);
293 setQuoteChar(quote);
294 }
295
296 /**
297 * Constructs a tokenizer splitting on the specified string.
298 *
299 * @param input The string which is to be parsed, not cloned.
300 * @param delim The field delimiter string.
301 */
302 public StrTokenizer(final char[] input, final String delim) {
303 this(input);
304 setDelimiterString(delim);
305 }
306
307 /**
308 * Constructs a tokenizer splitting using the specified delimiter matcher.
309 *
310 * @param input The string which is to be parsed, not cloned.
311 * @param delim The field delimiter matcher.
312 */
313 public StrTokenizer(final char[] input, final StrMatcher delim) {
314 this(input);
315 setDelimiterMatcher(delim);
316 }
317
318 /**
319 * Constructs a tokenizer splitting using the specified delimiter matcher
320 * and handling quotes using the specified quote matcher.
321 *
322 * @param input The string which is to be parsed, not cloned.
323 * @param delim The field delimiter character.
324 * @param quote The field quoted string character.
325 */
326 public StrTokenizer(final char[] input, final StrMatcher delim, final StrMatcher quote) {
327 this(input, delim);
328 setQuoteMatcher(quote);
329 }
330
331 /**
332 * Constructs a tokenizer splitting on space, tab, newline and formfeed
333 * as per StringTokenizer.
334 *
335 * @param input The string which is to be parsed.
336 */
337 public StrTokenizer(final String input) {
338 if (input != null) {
339 chars = input.toCharArray();
340 } else {
341 chars = null;
342 }
343 }
344
345 /**
346 * Constructs a tokenizer splitting on the specified delimiter character.
347 *
348 * @param input The string which is to be parsed.
349 * @param delim The field delimiter character.
350 */
351 public StrTokenizer(final String input, final char delim) {
352 this(input);
353 setDelimiterChar(delim);
354 }
355
356 /**
357 * Constructs a tokenizer splitting on the specified delimiter character
358 * and handling quotes using the specified quote character.
359 *
360 * @param input The string which is to be parsed.
361 * @param delim The field delimiter character.
362 * @param quote The field quoted string character.
363 */
364 public StrTokenizer(final String input, final char delim, final char quote) {
365 this(input, delim);
366 setQuoteChar(quote);
367 }
368
369 /**
370 * Constructs a tokenizer splitting on the specified delimiter string.
371 *
372 * @param input The string which is to be parsed.
373 * @param delim The field delimiter string.
374 */
375 public StrTokenizer(final String input, final String delim) {
376 this(input);
377 setDelimiterString(delim);
378 }
379
380 /**
381 * Constructs a tokenizer splitting using the specified delimiter matcher.
382 *
383 * @param input The string which is to be parsed.
384 * @param delim The field delimiter matcher.
385 */
386 public StrTokenizer(final String input, final StrMatcher delim) {
387 this(input);
388 setDelimiterMatcher(delim);
389 }
390
391 /**
392 * Constructs a tokenizer splitting using the specified delimiter matcher
393 * and handling quotes using the specified quote matcher.
394 *
395 * @param input The string which is to be parsed.
396 * @param delim The field delimiter matcher.
397 * @param quote The field quoted string matcher.
398 */
399 public StrTokenizer(final String input, final StrMatcher delim, final StrMatcher quote) {
400 this(input, delim);
401 setQuoteMatcher(quote);
402 }
403
404 /**
405 * Always throws {@link UnsupportedOperationException}.
406 *
407 * @param obj this parameter ignored.
408 * @throws UnsupportedOperationException Thrown because this operation is unsupported.
409 */
410 @Override
411 public void add(final String obj) {
412 throw new UnsupportedOperationException("add() is unsupported");
413 }
414
415 /**
416 * Adds a token to a list, paying attention to the parameters we've set.
417 *
418 * @param list The list to add to.
419 * @param tok The token to add.
420 */
421 private void addToken(final List<String> list, String tok) {
422 if (StringUtils.isEmpty(tok)) {
423 if (isIgnoreEmptyTokens()) {
424 return;
425 }
426 if (isEmptyTokenAsNull()) {
427 tok = null;
428 }
429 }
430 list.add(tok);
431 }
432
433 /**
434 * Checks if tokenization has been done, and if not then do it.
435 */
436 private void checkTokenized() {
437 if (tokens == null) {
438 if (chars == null) {
439 // still call tokenize as subclass may do some work
440 final List<String> split = tokenize(null, 0, 0);
441 tokens = split.toArray(ArrayUtils.EMPTY_STRING_ARRAY);
442 } else {
443 final List<String> split = tokenize(chars, 0, chars.length);
444 tokens = split.toArray(ArrayUtils.EMPTY_STRING_ARRAY);
445 }
446 }
447 }
448
449 /**
450 * Creates a new instance of this Tokenizer. The new instance is reset so
451 * that it will be at the start of the token list.
452 * If a {@link CloneNotSupportedException} is caught, return {@code null}.
453 *
454 * @return A new instance of this Tokenizer which has been reset.
455 */
456 @Override
457 public Object clone() {
458 try {
459 return cloneReset();
460 } catch (final CloneNotSupportedException ex) {
461 return null;
462 }
463 }
464
465 /**
466 * Creates a new instance of this Tokenizer. The new instance is reset so that
467 * it will be at the start of the token list.
468 *
469 * @return A new instance of this Tokenizer which has been reset.
470 * @throws CloneNotSupportedException Thrown if there is a problem cloning.
471 */
472 Object cloneReset() throws CloneNotSupportedException {
473 // this method exists to enable 100% test coverage
474 final StrTokenizer cloned = (StrTokenizer) super.clone();
475 if (cloned.chars != null) {
476 cloned.chars = cloned.chars.clone();
477 }
478 cloned.reset();
479 return cloned;
480 }
481
482 /**
483 * Gets the String content that the tokenizer is parsing.
484 *
485 * @return The string content being parsed.
486 */
487 public String getContent() {
488 if (chars == null) {
489 return null;
490 }
491 return new String(chars);
492 }
493
494 /**
495 * Gets the field delimiter matcher.
496 *
497 * @return The delimiter matcher in use.
498 */
499 public StrMatcher getDelimiterMatcher() {
500 return this.delimMatcher;
501 }
502
503 /**
504 * Gets the ignored character matcher.
505 * <p>
506 * These characters are ignored when parsing the String, unless they are
507 * within a quoted region.
508 * The default value is not to ignore anything.
509 * </p>
510 *
511 * @return The ignored matcher in use.
512 */
513 public StrMatcher getIgnoredMatcher() {
514 return ignoredMatcher;
515 }
516
517 /**
518 * Gets the quote matcher currently in use.
519 * <p>
520 * The quote character is used to wrap data between the tokens.
521 * This enables delimiters to be entered as data.
522 * The default value is '"' (double quote).
523 * </p>
524 *
525 * @return The quote matcher in use.
526 */
527 public StrMatcher getQuoteMatcher() {
528 return quoteMatcher;
529 }
530
531 /**
532 * Gets a copy of the full token list as an independent modifiable array.
533 *
534 * @return The tokens as a String array.
535 */
536 public String[] getTokenArray() {
537 checkTokenized();
538 return tokens.clone();
539 }
540
541 /**
542 * Gets a copy of the full token list as an independent modifiable list.
543 *
544 * @return The tokens as a String array.
545 */
546 public List<String> getTokenList() {
547 checkTokenized();
548 final List<String> list = new ArrayList<>(tokens.length);
549 list.addAll(Arrays.asList(tokens));
550 return list;
551 }
552
553 /**
554 * Gets the trimmer character matcher.
555 * <p>
556 * These characters are trimmed off on each side of the delimiter
557 * until the token or quote is found.
558 * The default value is not to trim anything.
559 * </p>
560 *
561 * @return The trimmer matcher in use.
562 */
563 public StrMatcher getTrimmerMatcher() {
564 return trimmerMatcher;
565 }
566
567 /**
568 * Tests whether there are any more tokens.
569 *
570 * @return true if there are more tokens.
571 */
572 @Override
573 public boolean hasNext() {
574 checkTokenized();
575 return tokenPos < tokens.length;
576 }
577
578 /**
579 * Tests whether there are any previous tokens that can be iterated to.
580 *
581 * @return true if there are previous tokens.
582 */
583 @Override
584 public boolean hasPrevious() {
585 checkTokenized();
586 return tokenPos > 0;
587 }
588
589 /**
590 * Tests whether the tokenizer currently returns empty tokens as null. The default for this property is false.
591 *
592 * @return true if empty tokens are returned as null.
593 */
594 public boolean isEmptyTokenAsNull() {
595 return this.emptyAsNull;
596 }
597
598 /**
599 * Tests whether the tokenizer currently ignores empty tokens. The default for this property is true.
600 *
601 * @return true if empty tokens are not returned.
602 */
603 public boolean isIgnoreEmptyTokens() {
604 return ignoreEmptyTokens;
605 }
606
607 /**
608 * Tests whether the characters at the index specified match the quote already matched in readNextToken().
609 *
610 * @param srcChars The character array being tokenized.
611 * @param pos The position to check for a quote.
612 * @param len The length of the character array being tokenized.
613 * @param quoteStart The start position of the matched quote, 0 if no quoting.
614 * @param quoteLen The length of the matched quote, 0 if no quoting.
615 * @return true if a quote is matched.
616 */
617 private boolean isQuote(final char[] srcChars, final int pos, final int len, final int quoteStart, final int quoteLen) {
618 for (int i = 0; i < quoteLen; i++) {
619 if (pos + i >= len || srcChars[pos + i] != srcChars[quoteStart + i]) {
620 return false;
621 }
622 }
623 return true;
624 }
625
626 /**
627 * Gets the next token.
628 *
629 * @return The next String token.
630 * @throws NoSuchElementException Thrown if there are no more elements.
631 */
632 @Override
633 public String next() {
634 if (hasNext()) {
635 return tokens[tokenPos++];
636 }
637 throw new NoSuchElementException();
638 }
639
640 /**
641 * Gets the index of the next token to return.
642 *
643 * @return The next token index.
644 */
645 @Override
646 public int nextIndex() {
647 return tokenPos;
648 }
649
650 /**
651 * Gets the next token from the String.
652 * Equivalent to {@link #next()} except it returns null rather than
653 * throwing {@link NoSuchElementException} when no tokens remain.
654 *
655 * @return The next sequential token, or null when no more tokens are found.
656 */
657 public String nextToken() {
658 if (hasNext()) {
659 return tokens[tokenPos++];
660 }
661 return null;
662 }
663
664 /**
665 * Gets the token previous to the last returned token.
666 *
667 * @return The previous token.
668 */
669 @Override
670 public String previous() {
671 if (hasPrevious()) {
672 return tokens[--tokenPos];
673 }
674 throw new NoSuchElementException();
675 }
676
677 /**
678 * Gets the index of the previous token.
679 *
680 * @return The previous token index.
681 */
682 @Override
683 public int previousIndex() {
684 return tokenPos - 1;
685 }
686
687 /**
688 * Gets the previous token from the String.
689 *
690 * @return The previous sequential token, or null when no more tokens are found.
691 */
692 public String previousToken() {
693 if (hasPrevious()) {
694 return tokens[--tokenPos];
695 }
696 return null;
697 }
698
699 /**
700 * Reads character by character through the String to get the next token.
701 *
702 * @param srcChars The character array being tokenized.
703 * @param start The first character of field.
704 * @param len The length of the character array being tokenized.
705 * @param workArea A temporary work area.
706 * @param tokenList The list of parsed tokens.
707 * @return The starting position of the next field (the character
708 * immediately after the delimiter), or -1 if end of string found.
709 */
710 private int readNextToken(final char[] srcChars, int start, final int len, final StrBuilder workArea, final List<String> tokenList) {
711 // skip all leading whitespace, unless it is the
712 // field delimiter or the quote character
713 while (start < len) {
714 final int removeLen = Math.max(
715 getIgnoredMatcher().isMatch(srcChars, start, start, len),
716 getTrimmerMatcher().isMatch(srcChars, start, start, len));
717 if (removeLen == 0 ||
718 getDelimiterMatcher().isMatch(srcChars, start, start, len) > 0 ||
719 getQuoteMatcher().isMatch(srcChars, start, start, len) > 0) {
720 break;
721 }
722 start += removeLen;
723 }
724
725 // handle reaching end
726 if (start >= len) {
727 addToken(tokenList, StringUtils.EMPTY);
728 return -1;
729 }
730
731 // handle empty token
732 final int delimLen = getDelimiterMatcher().isMatch(srcChars, start, start, len);
733 if (delimLen > 0) {
734 addToken(tokenList, StringUtils.EMPTY);
735 return start + delimLen;
736 }
737
738 // handle found token
739 final int quoteLen = getQuoteMatcher().isMatch(srcChars, start, start, len);
740 if (quoteLen > 0) {
741 return readWithQuotes(srcChars, start + quoteLen, len, workArea, tokenList, start, quoteLen);
742 }
743 return readWithQuotes(srcChars, start, len, workArea, tokenList, 0, 0);
744 }
745
746 /**
747 * Reads a possibly quoted string token.
748 *
749 * @param srcChars The character array being tokenized.
750 * @param start The first character of field.
751 * @param len The length of the character array being tokenized.
752 * @param workArea A temporary work area.
753 * @param tokenList The list of parsed tokens.
754 * @param quoteStart The start position of the matched quote, 0 if no quoting.
755 * @param quoteLen The length of the matched quote, 0 if no quoting.
756 * @return The starting position of the next field (the character
757 * immediately after the delimiter, or if end of string found,
758 * then the length of string.
759 */
760 private int readWithQuotes(final char[] srcChars, final int start, final int len, final StrBuilder workArea,
761 final List<String> tokenList, final int quoteStart, final int quoteLen) {
762 // Loop until we've found the end of the quoted
763 // string or the end of the input
764 workArea.clear();
765 int pos = start;
766 boolean quoting = quoteLen > 0;
767 int trimStart = 0;
768
769 while (pos < len) {
770 // quoting mode can occur several times throughout a string
771 // we must switch between quoting and non-quoting until we
772 // encounter a non-quoted delimiter, or end of string
773 if (quoting) {
774 // In quoting mode
775
776 // If we've found a quote character, see if it's
777 // followed by a second quote. If so, then we need
778 // to actually put the quote character into the token
779 // rather than end the token.
780 if (isQuote(srcChars, pos, len, quoteStart, quoteLen)) {
781 if (isQuote(srcChars, pos + quoteLen, len, quoteStart, quoteLen)) {
782 // matched pair of quotes, thus an escaped quote
783 workArea.append(srcChars, pos, quoteLen);
784 pos += quoteLen * 2;
785 trimStart = workArea.size();
786 continue;
787 }
788
789 // end of quoting
790 quoting = false;
791 pos += quoteLen;
792 continue;
793 }
794
795 } else {
796 // Not in quoting mode
797
798 // check for delimiter, and thus end of token
799 final int delimLen = getDelimiterMatcher().isMatch(srcChars, pos, start, len);
800 if (delimLen > 0) {
801 // return condition when end of token found
802 addToken(tokenList, workArea.substring(0, trimStart));
803 return pos + delimLen;
804 }
805
806 // check for quote, and thus back into quoting mode
807 if (quoteLen > 0 && isQuote(srcChars, pos, len, quoteStart, quoteLen)) {
808 quoting = true;
809 pos += quoteLen;
810 continue;
811 }
812
813 // check for ignored (outside quotes), and ignore
814 final int ignoredLen = getIgnoredMatcher().isMatch(srcChars, pos, start, len);
815 if (ignoredLen > 0) {
816 pos += ignoredLen;
817 continue;
818 }
819
820 // check for trimmed character
821 // don't yet know if it's at the end, so copy to workArea
822 // use trimStart to keep track of trim at the end
823 final int trimmedLen = getTrimmerMatcher().isMatch(srcChars, pos, start, len);
824 if (trimmedLen > 0) {
825 workArea.append(srcChars, pos, trimmedLen);
826 pos += trimmedLen;
827 continue;
828 }
829 }
830 // copy regular character from inside quotes
831 workArea.append(srcChars[pos++]);
832 trimStart = workArea.size();
833 }
834
835 // return condition when end of string found
836 addToken(tokenList, workArea.substring(0, trimStart));
837 return -1;
838 }
839
840 /**
841 * Always throws {@link UnsupportedOperationException}.
842 *
843 * @throws UnsupportedOperationException Thrown because this operation is unsupported.
844 */
845 @Override
846 public void remove() {
847 throw new UnsupportedOperationException("remove() is unsupported");
848 }
849
850 /**
851 * Resets this tokenizer, forgetting all parsing and iteration already completed.
852 * <p>
853 * This method allows the same tokenizer to be reused for the same String.
854 * </p>
855 *
856 * @return {@code this} instance.
857 */
858 public StrTokenizer reset() {
859 tokenPos = 0;
860 tokens = null;
861 return this;
862 }
863
864 /**
865 * Reset this tokenizer, giving it a new input string to parse.
866 * In this manner you can re-use a tokenizer with the same settings
867 * on multiple input lines.
868 *
869 * @param input The new character array to tokenize, not cloned, null sets no text to parse.
870 * @return {@code this} instance.
871 */
872 public StrTokenizer reset(final char[] input) {
873 reset();
874 this.chars = ArrayUtils.clone(input);
875 return this;
876 }
877
878 /**
879 * Reset this tokenizer, giving it a new input string to parse.
880 * In this manner you can re-use a tokenizer with the same settings
881 * on multiple input lines.
882 *
883 * @param input The new string to tokenize, null sets no text to parse.
884 * @return {@code this} instance.
885 */
886 public StrTokenizer reset(final String input) {
887 reset();
888 if (input != null) {
889 this.chars = input.toCharArray();
890 } else {
891 this.chars = null;
892 }
893 return this;
894 }
895
896 /**
897 * Sets no token and always throws {@link UnsupportedOperationException}.
898 *
899 * @param obj this parameter ignored.
900 * @throws UnsupportedOperationException Thrown because this operation is unsupported.
901 */
902 @Override
903 public void set(final String obj) {
904 throw new UnsupportedOperationException("set() is unsupported");
905 }
906
907 /**
908 * Sets the field delimiter character.
909 *
910 * @param delim The delimiter character to use.
911 * @return {@code this} instance.
912 */
913 public StrTokenizer setDelimiterChar(final char delim) {
914 return setDelimiterMatcher(StrMatcher.charMatcher(delim));
915 }
916
917 /**
918 * Sets the field delimiter matcher.
919 * <p>
920 * The delimiter is used to separate one token from another.
921 * </p>
922 *
923 * @param delim The delimiter matcher to use.
924 * @return {@code this} instance.
925 */
926 public StrTokenizer setDelimiterMatcher(final StrMatcher delim) {
927 if (delim == null) {
928 this.delimMatcher = StrMatcher.noneMatcher();
929 } else {
930 this.delimMatcher = delim;
931 }
932 return this;
933 }
934
935 /**
936 * Sets the field delimiter string.
937 *
938 * @param delim The delimiter string to use.
939 * @return {@code this} instance.
940 */
941 public StrTokenizer setDelimiterString(final String delim) {
942 return setDelimiterMatcher(StrMatcher.stringMatcher(delim));
943 }
944
945 /**
946 * Sets whether the tokenizer should return empty tokens as null.
947 * The default for this property is false.
948 *
949 * @param emptyAsNull whether empty tokens are returned as null.
950 * @return {@code this} instance.
951 */
952 public StrTokenizer setEmptyTokenAsNull(final boolean emptyAsNull) {
953 this.emptyAsNull = emptyAsNull;
954 return this;
955 }
956
957 /**
958 * Sets the character to ignore.
959 * <p>
960 * This character is ignored when parsing the String, unless it is
961 * within a quoted region.
962 *
963 * @param ignored The ignored character to use.
964 * @return {@code this} instance.
965 */
966 public StrTokenizer setIgnoredChar(final char ignored) {
967 return setIgnoredMatcher(StrMatcher.charMatcher(ignored));
968 }
969
970 /**
971 * Sets the matcher for characters to ignore.
972 * <p>
973 * These characters are ignored when parsing the String, unless they are
974 * within a quoted region.
975 * </p>
976 *
977 * @param ignored The ignored matcher to use, null ignored.
978 * @return {@code this} instance.
979 */
980 public StrTokenizer setIgnoredMatcher(final StrMatcher ignored) {
981 if (ignored != null) {
982 this.ignoredMatcher = ignored;
983 }
984 return this;
985 }
986
987 /**
988 * Sets whether the tokenizer should ignore and not return empty tokens.
989 * The default for this property is true.
990 *
991 * @param ignoreEmptyTokens whether empty tokens are not returned.
992 * @return {@code this} instance.
993 */
994 public StrTokenizer setIgnoreEmptyTokens(final boolean ignoreEmptyTokens) {
995 this.ignoreEmptyTokens = ignoreEmptyTokens;
996 return this;
997 }
998
999 /**
1000 * Sets the quote character to use.
1001 * <p>
1002 * The quote character is used to wrap data between the tokens.
1003 * This enables delimiters to be entered as data.
1004 * </p>
1005 *
1006 * @param quote The quote character to use.
1007 * @return {@code this} instance.
1008 */
1009 public StrTokenizer setQuoteChar(final char quote) {
1010 return setQuoteMatcher(StrMatcher.charMatcher(quote));
1011 }
1012
1013 /**
1014 * Sets the quote matcher to use.
1015 * <p>
1016 * The quote character is used to wrap data between the tokens.
1017 * This enables delimiters to be entered as data.
1018 * </p>
1019 *
1020 * @param quote The quote matcher to use, null ignored.
1021 * @return {@code this} instance.
1022 */
1023 public StrTokenizer setQuoteMatcher(final StrMatcher quote) {
1024 if (quote != null) {
1025 this.quoteMatcher = quote;
1026 }
1027 return this;
1028 }
1029
1030 /**
1031 * Sets the matcher for characters to trim.
1032 * <p>
1033 * These characters are trimmed off on each side of the delimiter
1034 * until the token or quote is found.
1035 * </p>
1036 *
1037 * @param trimmer The trimmer matcher to use, null ignored.
1038 * @return {@code this} instance.
1039 */
1040 public StrTokenizer setTrimmerMatcher(final StrMatcher trimmer) {
1041 if (trimmer != null) {
1042 this.trimmerMatcher = trimmer;
1043 }
1044 return this;
1045 }
1046
1047 /**
1048 * Gets the number of tokens found in the String.
1049 *
1050 * @return The number of matched tokens.
1051 */
1052 public int size() {
1053 checkTokenized();
1054 return tokens.length;
1055 }
1056
1057 /**
1058 * Internal method to performs the tokenization.
1059 * <p>
1060 * Most users of this class do not need to call this method. This method
1061 * will be called automatically by other (public) methods when required.
1062 * </p>
1063 * <p>
1064 * This method exists to allow subclasses to add code before or after the
1065 * tokenization. For example, a subclass could alter the character array,
1066 * offset or count to be parsed, or call the tokenizer multiple times on
1067 * multiple strings. It is also be possible to filter the results.
1068 * </p>
1069 * <p>
1070 * {@link StrTokenizer} will always pass a zero offset and a count
1071 * equal to the length of the array to this method, however a subclass
1072 * may pass other values, or even an entirely different array.
1073 * </p>
1074 *
1075 * @param srcChars The character array being tokenized, may be null.
1076 * @param offset The start position within the character array, must be valid.
1077 * @param count The number of characters to tokenize, must be valid.
1078 * @return The modifiable list of String tokens, unmodifiable if null array or zero count.
1079 */
1080 protected List<String> tokenize(final char[] srcChars, final int offset, final int count) {
1081 if (ArrayUtils.isEmpty(srcChars)) {
1082 return Collections.emptyList();
1083 }
1084 final StrBuilder buf = new StrBuilder();
1085 final List<String> tokenList = new ArrayList<>();
1086 int pos = offset;
1087
1088 // loop around the entire buffer
1089 while (pos >= 0 && pos < count) {
1090 // find next token
1091 pos = readNextToken(srcChars, pos, count, buf, tokenList);
1092
1093 // handle case where end of string is a delimiter
1094 if (pos >= count) {
1095 addToken(tokenList, StringUtils.EMPTY);
1096 }
1097 }
1098 return tokenList;
1099 }
1100
1101 /**
1102 * Gets the String content that the tokenizer is parsing.
1103 *
1104 * @return The string content being parsed.
1105 */
1106 @Override
1107 public String toString() {
1108 if (tokens == null) {
1109 return "StrTokenizer[not tokenized yet]";
1110 }
1111 return "StrTokenizer" + getTokenList();
1112 }
1113
1114 }