001/*
002 * Licensed to the Apache Software Foundation (ASF) under one or more
003 * contributor license agreements.  See the NOTICE file distributed with
004 * this work for additional information regarding copyright ownership.
005 * The ASF licenses this file to You under the Apache License, Version 2.0
006 * (the "License"); you may not use this file except in compliance with
007 * the License.  You may obtain a copy of the License at
008 *
009 *      https://www.apache.org/licenses/LICENSE-2.0
010 *
011 * Unless required by applicable law or agreed to in writing, software
012 * distributed under the License is distributed on an "AS IS" BASIS,
013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
014 * See the License for the specific language governing permissions and
015 * limitations under the License.
016 */
017package org.apache.commons.lang3.text;
018
019import java.util.ArrayList;
020import java.util.Arrays;
021import java.util.Collections;
022import java.util.List;
023import java.util.ListIterator;
024import java.util.NoSuchElementException;
025import java.util.StringTokenizer;
026
027import org.apache.commons.lang3.ArrayUtils;
028import org.apache.commons.lang3.StringUtils;
029
030/**
031 * Tokenizes a string based on delimiters (separators)
032 * and supporting quoting and ignored character concepts.
033 * <p>
034 * This class can split a String into many smaller strings. It aims
035 * to do a similar job to {@link java.util.StringTokenizer StringTokenizer},
036 * however it offers much more control and flexibility including implementing
037 * the {@link ListIterator} interface. By default, it is set up
038 * like {@link StringTokenizer}.
039 * </p>
040 * <p>
041 * The input String is split into a number of <em>tokens</em>.
042 * Each token is separated from the next String by a <em>delimiter</em>.
043 * One or more delimiter characters must be specified.
044 * </p>
045 * <p>
046 * Each token may be surrounded by quotes.
047 * The <em>quote</em> matcher specifies the quote character(s).
048 * A quote may be escaped within a quoted section by duplicating itself.
049 * </p>
050 * <p>
051 * Between each token and the delimiter are potentially characters that need trimming.
052 * The <em>trimmer</em> matcher specifies these characters.
053 * One usage might be to trim whitespace characters.
054 * </p>
055 * <p>
056 * At any point outside the quotes there might potentially be invalid characters.
057 * The <em>ignored</em> matcher specifies these characters to be removed.
058 * One usage might be to remove new line characters.
059 * </p>
060 * <p>
061 * Empty tokens may be removed or returned as null.
062 * </p>
063 * <pre>
064 * "a,b,c"         - Three tokens "a","b","c"   (comma delimiter)
065 * " a, b , c "    - Three tokens "a","b","c"   (default CSV processing trims whitespace)
066 * "a, ", b ,", c" - Three tokens "a, " , " b ", ", c" (quoted text untouched)
067 * </pre>
068 *
069 * <table>
070 *  <caption>StrTokenizer properties and options</caption>
071 *  <tr>
072 *   <th>Property</th><th>Type</th><th>Default</th>
073 *  </tr>
074 *  <tr>
075 *   <td>delim</td><td>CharSetMatcher</td><td>{ \t\n\r\f}</td>
076 *  </tr>
077 *  <tr>
078 *   <td>quote</td><td>NoneMatcher</td><td>{}</td>
079 *  </tr>
080 *  <tr>
081 *   <td>ignore</td><td>NoneMatcher</td><td>{}</td>
082 *  </tr>
083 *  <tr>
084 *   <td>emptyTokenAsNull</td><td>boolean</td><td>false</td>
085 *  </tr>
086 *  <tr>
087 *   <td>ignoreEmptyTokens</td><td>boolean</td><td>true</td>
088 *  </tr>
089 * </table>
090 *
091 * @since 2.2
092 * @deprecated As of <a href="https://commons.apache.org/proper/commons-lang/changes-report.html#a3.6">3.6</a>, use Apache Commons Text
093 * <a href="https://commons.apache.org/proper/commons-text/javadocs/api-release/org/apache/commons/text/StringTokenizer.html">
094 * StringTokenizer</a>.
095 */
096@Deprecated
097public class StrTokenizer implements ListIterator<String>, Cloneable {
098
099    // @formatter:off
100    private static final StrTokenizer CSV_TOKENIZER_PROTOTYPE = new StrTokenizer()
101            .setDelimiterMatcher(StrMatcher.commaMatcher())
102            .setQuoteMatcher(StrMatcher.doubleQuoteMatcher())
103            .setIgnoredMatcher(StrMatcher.noneMatcher())
104            .setTrimmerMatcher(StrMatcher.trimMatcher())
105            .setEmptyTokenAsNull(false)
106            .setIgnoreEmptyTokens(false);
107    // @formatter:on
108
109    // @formatter:off
110    private static final StrTokenizer TSV_TOKENIZER_PROTOTYPE = new StrTokenizer()
111            .setDelimiterMatcher(StrMatcher.tabMatcher())
112            .setQuoteMatcher(StrMatcher.doubleQuoteMatcher())
113            .setIgnoredMatcher(StrMatcher.noneMatcher())
114            .setTrimmerMatcher(StrMatcher.trimMatcher())
115            .setEmptyTokenAsNull(false)
116            .setIgnoreEmptyTokens(false);
117    // @formatter:on
118
119    /**
120     * Gets a clone of {@code CSV_TOKENIZER_PROTOTYPE}.
121     *
122     * @return A clone of {@code CSV_TOKENIZER_PROTOTYPE}.
123     */
124    private static StrTokenizer getCSVClone() {
125        return (StrTokenizer) CSV_TOKENIZER_PROTOTYPE.clone();
126    }
127
128    /**
129     * Gets a new tokenizer instance which parses Comma Separated Value strings
130     * initializing it with the given input.  The default for CSV processing
131     * will be trim whitespace from both ends (which can be overridden with
132     * the setTrimmer method).
133     * <p>
134     * You must call a "reset" method to set the string which you want to parse.
135     * </p>
136     *
137     * @return A new tokenizer instance which parses Comma Separated Value strings.
138     */
139    public static StrTokenizer getCSVInstance() {
140        return getCSVClone();
141    }
142
143    /**
144     * Gets a new tokenizer instance which parses Comma Separated Value strings
145     * initializing it with the given input.  The default for CSV processing
146     * will be trim whitespace from both ends (which can be overridden with
147     * the setTrimmer method).
148     *
149     * @param input  The text to parse.
150     * @return A new tokenizer instance which parses Comma Separated Value strings.
151     */
152    public static StrTokenizer getCSVInstance(final char[] input) {
153        final StrTokenizer tok = getCSVClone();
154        tok.reset(input);
155        return tok;
156    }
157
158    /**
159     * Gets a new tokenizer instance which parses Comma Separated Value strings
160     * initializing it with the given input.  The default for CSV processing
161     * will be trim whitespace from both ends (which can be overridden with
162     * the setTrimmer method).
163     *
164     * @param input  The text to parse.
165     * @return A new tokenizer instance which parses Comma Separated Value strings.
166     */
167    public static StrTokenizer getCSVInstance(final String input) {
168        final StrTokenizer tok = getCSVClone();
169        tok.reset(input);
170        return tok;
171    }
172
173    /**
174     * Gets a clone of {@code TSV_TOKENIZER_PROTOTYPE}.
175     *
176     * @return A clone of {@code TSV_TOKENIZER_PROTOTYPE}.
177     */
178    private static StrTokenizer getTSVClone() {
179        return (StrTokenizer) TSV_TOKENIZER_PROTOTYPE.clone();
180    }
181
182    /**
183     * Gets a new tokenizer instance which parses Tab Separated Value strings.
184     * The default for CSV processing will be trim whitespace from both ends
185     * (which can be overridden with the setTrimmer method).
186     * <p>
187     * You must call a "reset" method to set the string which you want to parse.
188     * </p>
189     *
190     * @return A new tokenizer instance which parses Tab Separated Value strings.
191     */
192    public static StrTokenizer getTSVInstance() {
193        return getTSVClone();
194    }
195
196    /**
197     * Gets a new tokenizer instance which parses Tab Separated Value strings.
198     * The default for CSV processing will be trim whitespace from both ends
199     * (which can be overridden with the setTrimmer method).
200     *
201     * @param input  The string to parse.
202     * @return A new tokenizer instance which parses Tab Separated Value strings.
203     */
204    public static StrTokenizer getTSVInstance(final char[] input) {
205        final StrTokenizer tok = getTSVClone();
206        tok.reset(input);
207        return tok;
208    }
209
210    /**
211     * Gets a new tokenizer instance which parses Tab Separated Value strings.
212     * The default for CSV processing will be trim whitespace from both ends
213     * (which can be overridden with the setTrimmer method).
214     *
215     * @param input  The string to parse.
216     * @return A new tokenizer instance which parses Tab Separated Value strings.
217     */
218    public static StrTokenizer getTSVInstance(final String input) {
219        final StrTokenizer tok = getTSVClone();
220        tok.reset(input);
221        return tok;
222    }
223
224    /** The text to work on. */
225    private char[] chars;
226
227    /** The parsed tokens */
228    private String[] tokens;
229
230    /** The current iteration position */
231    private int tokenPos;
232
233    /** The delimiter matcher */
234    private StrMatcher delimMatcher = StrMatcher.splitMatcher();
235
236    /** The quote matcher */
237    private StrMatcher quoteMatcher = StrMatcher.noneMatcher();
238
239    /** The ignored matcher */
240    private StrMatcher ignoredMatcher = StrMatcher.noneMatcher();
241
242    /** The trimmer matcher */
243    private StrMatcher trimmerMatcher = StrMatcher.noneMatcher();
244
245    /** Whether to return empty tokens as null */
246    private boolean emptyAsNull;
247
248    /** Whether to ignore empty tokens */
249    private boolean ignoreEmptyTokens = true;
250
251    /**
252     * Constructs a tokenizer splitting on space, tab, newline and formfeed
253     * as per StringTokenizer, but with no text to tokenize.
254     * <p>
255     * This constructor is normally used with {@link #reset(String)}.
256     * </p>
257     */
258    public StrTokenizer() {
259        this.chars = null;
260    }
261
262    /**
263     * Constructs a tokenizer splitting on space, tab, newline and formfeed
264     * as per StringTokenizer.
265     *
266     * @param input  The string which is to be parsed, not cloned.
267     */
268    public StrTokenizer(final char[] input) {
269        this.chars = ArrayUtils.clone(input);
270    }
271
272    /**
273     * Constructs a tokenizer splitting on the specified character.
274     *
275     * @param input  The string which is to be parsed, not cloned.
276     * @param delim The field delimiter character.
277     */
278    public StrTokenizer(final char[] input, final char delim) {
279        this(input);
280        setDelimiterChar(delim);
281    }
282
283    /**
284     * Constructs a tokenizer splitting on the specified delimiter character
285     * and handling quotes using the specified quote character.
286     *
287     * @param input  The string which is to be parsed, not cloned.
288     * @param delim  The field delimiter character.
289     * @param quote  The field quoted string character.
290     */
291    public StrTokenizer(final char[] input, final char delim, final char quote) {
292        this(input, delim);
293        setQuoteChar(quote);
294    }
295
296    /**
297     * Constructs a tokenizer splitting on the specified string.
298     *
299     * @param input  The string which is to be parsed, not cloned.
300     * @param delim The field delimiter string.
301     */
302    public StrTokenizer(final char[] input, final String delim) {
303        this(input);
304        setDelimiterString(delim);
305    }
306
307    /**
308     * Constructs a tokenizer splitting using the specified delimiter matcher.
309     *
310     * @param input  The string which is to be parsed, not cloned.
311     * @param delim  The field delimiter matcher.
312     */
313    public StrTokenizer(final char[] input, final StrMatcher delim) {
314        this(input);
315        setDelimiterMatcher(delim);
316    }
317
318    /**
319     * Constructs a tokenizer splitting using the specified delimiter matcher
320     * and handling quotes using the specified quote matcher.
321     *
322     * @param input  The string which is to be parsed, not cloned.
323     * @param delim  The field delimiter character.
324     * @param quote  The field quoted string character.
325     */
326    public StrTokenizer(final char[] input, final StrMatcher delim, final StrMatcher quote) {
327        this(input, delim);
328        setQuoteMatcher(quote);
329    }
330
331    /**
332     * Constructs a tokenizer splitting on space, tab, newline and formfeed
333     * as per StringTokenizer.
334     *
335     * @param input  The string which is to be parsed.
336     */
337    public StrTokenizer(final String input) {
338        if (input != null) {
339            chars = input.toCharArray();
340        } else {
341            chars = null;
342        }
343    }
344
345    /**
346     * Constructs a tokenizer splitting on the specified delimiter character.
347     *
348     * @param input  The string which is to be parsed.
349     * @param delim  The field delimiter character.
350     */
351    public StrTokenizer(final String input, final char delim) {
352        this(input);
353        setDelimiterChar(delim);
354    }
355
356    /**
357     * Constructs a tokenizer splitting on the specified delimiter character
358     * and handling quotes using the specified quote character.
359     *
360     * @param input  The string which is to be parsed.
361     * @param delim  The field delimiter character.
362     * @param quote  The field quoted string character.
363     */
364    public StrTokenizer(final String input, final char delim, final char quote) {
365        this(input, delim);
366        setQuoteChar(quote);
367    }
368
369    /**
370     * Constructs a tokenizer splitting on the specified delimiter string.
371     *
372     * @param input  The string which is to be parsed.
373     * @param delim  The field delimiter string.
374     */
375    public StrTokenizer(final String input, final String delim) {
376        this(input);
377        setDelimiterString(delim);
378    }
379
380    /**
381     * Constructs a tokenizer splitting using the specified delimiter matcher.
382     *
383     * @param input  The string which is to be parsed.
384     * @param delim  The field delimiter matcher.
385     */
386    public StrTokenizer(final String input, final StrMatcher delim) {
387        this(input);
388        setDelimiterMatcher(delim);
389    }
390
391    /**
392     * Constructs a tokenizer splitting using the specified delimiter matcher
393     * and handling quotes using the specified quote matcher.
394     *
395     * @param input  The string which is to be parsed.
396     * @param delim  The field delimiter matcher.
397     * @param quote  The field quoted string matcher.
398     */
399    public StrTokenizer(final String input, final StrMatcher delim, final StrMatcher quote) {
400        this(input, delim);
401        setQuoteMatcher(quote);
402    }
403
404    /**
405     * Always throws {@link UnsupportedOperationException}.
406     *
407     * @param obj this parameter ignored.
408     * @throws UnsupportedOperationException Thrown because this operation is unsupported.
409     */
410    @Override
411    public void add(final String obj) {
412        throw new UnsupportedOperationException("add() is unsupported");
413    }
414
415    /**
416     * Adds a token to a list, paying attention to the parameters we've set.
417     *
418     * @param list  The list to add to.
419     * @param tok  The token to add.
420     */
421    private void addToken(final List<String> list, String tok) {
422        if (StringUtils.isEmpty(tok)) {
423            if (isIgnoreEmptyTokens()) {
424                return;
425            }
426            if (isEmptyTokenAsNull()) {
427                tok = null;
428            }
429        }
430        list.add(tok);
431    }
432
433    /**
434     * Checks if tokenization has been done, and if not then do it.
435     */
436    private void checkTokenized() {
437        if (tokens == null) {
438            if (chars == null) {
439                // still call tokenize as subclass may do some work
440                final List<String> split = tokenize(null, 0, 0);
441                tokens = split.toArray(ArrayUtils.EMPTY_STRING_ARRAY);
442            } else {
443                final List<String> split = tokenize(chars, 0, chars.length);
444                tokens = split.toArray(ArrayUtils.EMPTY_STRING_ARRAY);
445            }
446        }
447    }
448
449    /**
450     * Creates a new instance of this Tokenizer. The new instance is reset so
451     * that it will be at the start of the token list.
452     * If a {@link CloneNotSupportedException} is caught, return {@code null}.
453     *
454     * @return A new instance of this Tokenizer which has been reset.
455     */
456    @Override
457    public Object clone() {
458        try {
459            return cloneReset();
460        } catch (final CloneNotSupportedException ex) {
461            return null;
462        }
463    }
464
465    /**
466     * Creates a new instance of this Tokenizer. The new instance is reset so that
467     * it will be at the start of the token list.
468     *
469     * @return A new instance of this Tokenizer which has been reset.
470     * @throws CloneNotSupportedException Thrown if there is a problem cloning.
471     */
472    Object cloneReset() throws CloneNotSupportedException {
473        // this method exists to enable 100% test coverage
474        final StrTokenizer cloned = (StrTokenizer) super.clone();
475        if (cloned.chars != null) {
476            cloned.chars = cloned.chars.clone();
477        }
478        cloned.reset();
479        return cloned;
480    }
481
482    /**
483     * Gets the String content that the tokenizer is parsing.
484     *
485     * @return The string content being parsed.
486     */
487    public String getContent() {
488        if (chars == null) {
489            return null;
490        }
491        return new String(chars);
492    }
493
494    /**
495     * Gets the field delimiter matcher.
496     *
497     * @return The delimiter matcher in use.
498     */
499    public StrMatcher getDelimiterMatcher() {
500        return this.delimMatcher;
501    }
502
503    /**
504     * Gets the ignored character matcher.
505     * <p>
506     * These characters are ignored when parsing the String, unless they are
507     * within a quoted region.
508     * The default value is not to ignore anything.
509     * </p>
510     *
511     * @return The ignored matcher in use.
512     */
513    public StrMatcher getIgnoredMatcher() {
514        return ignoredMatcher;
515    }
516
517    /**
518     * Gets the quote matcher currently in use.
519     * <p>
520     * The quote character is used to wrap data between the tokens.
521     * This enables delimiters to be entered as data.
522     * The default value is '"' (double quote).
523     * </p>
524     *
525     * @return The quote matcher in use.
526     */
527    public StrMatcher getQuoteMatcher() {
528        return quoteMatcher;
529    }
530
531    /**
532     * Gets a copy of the full token list as an independent modifiable array.
533     *
534     * @return The tokens as a String array.
535     */
536    public String[] getTokenArray() {
537        checkTokenized();
538        return tokens.clone();
539    }
540
541    /**
542     * Gets a copy of the full token list as an independent modifiable list.
543     *
544     * @return The tokens as a String array.
545     */
546    public List<String> getTokenList() {
547        checkTokenized();
548        final List<String> list = new ArrayList<>(tokens.length);
549        list.addAll(Arrays.asList(tokens));
550        return list;
551    }
552
553    /**
554     * Gets the trimmer character matcher.
555     * <p>
556     * These characters are trimmed off on each side of the delimiter
557     * until the token or quote is found.
558     * The default value is not to trim anything.
559     * </p>
560     *
561     * @return The trimmer matcher in use.
562     */
563    public StrMatcher getTrimmerMatcher() {
564        return trimmerMatcher;
565    }
566
567    /**
568     * Tests whether there are any more tokens.
569     *
570     * @return true if there are more tokens.
571     */
572    @Override
573    public boolean hasNext() {
574        checkTokenized();
575        return tokenPos < tokens.length;
576    }
577
578    /**
579     * Tests whether there are any previous tokens that can be iterated to.
580     *
581     * @return true if there are previous tokens.
582     */
583    @Override
584    public boolean hasPrevious() {
585        checkTokenized();
586        return tokenPos > 0;
587    }
588
589    /**
590     * Tests whether the tokenizer currently returns empty tokens as null. The default for this property is false.
591     *
592     * @return true if empty tokens are returned as null.
593     */
594    public boolean isEmptyTokenAsNull() {
595        return this.emptyAsNull;
596    }
597
598    /**
599     * Tests whether the tokenizer currently ignores empty tokens. The default for this property is true.
600     *
601     * @return true if empty tokens are not returned.
602     */
603    public boolean isIgnoreEmptyTokens() {
604        return ignoreEmptyTokens;
605    }
606
607    /**
608     * Tests whether the characters at the index specified match the quote already matched in readNextToken().
609     *
610     * @param srcChars  The character array being tokenized.
611     * @param pos  The position to check for a quote.
612     * @param len  The length of the character array being tokenized.
613     * @param quoteStart  The start position of the matched quote, 0 if no quoting.
614     * @param quoteLen  The length of the matched quote, 0 if no quoting.
615     * @return true if a quote is matched.
616     */
617    private boolean isQuote(final char[] srcChars, final int pos, final int len, final int quoteStart, final int quoteLen) {
618        for (int i = 0; i < quoteLen; i++) {
619            if (pos + i >= len || srcChars[pos + i] != srcChars[quoteStart + i]) {
620                return false;
621            }
622        }
623        return true;
624    }
625
626    /**
627     * Gets the next token.
628     *
629     * @return The next String token.
630     * @throws NoSuchElementException Thrown if there are no more elements.
631     */
632    @Override
633    public String next() {
634        if (hasNext()) {
635            return tokens[tokenPos++];
636        }
637        throw new NoSuchElementException();
638    }
639
640    /**
641     * Gets the index of the next token to return.
642     *
643     * @return The next token index.
644     */
645    @Override
646    public int nextIndex() {
647        return tokenPos;
648    }
649
650    /**
651     * Gets the next token from the String.
652     * Equivalent to {@link #next()} except it returns null rather than
653     * throwing {@link NoSuchElementException} when no tokens remain.
654     *
655     * @return The next sequential token, or null when no more tokens are found.
656     */
657    public String nextToken() {
658        if (hasNext()) {
659            return tokens[tokenPos++];
660        }
661        return null;
662    }
663
664    /**
665     * Gets the token previous to the last returned token.
666     *
667     * @return The previous token.
668     */
669    @Override
670    public String previous() {
671        if (hasPrevious()) {
672            return tokens[--tokenPos];
673        }
674        throw new NoSuchElementException();
675    }
676
677    /**
678     * Gets the index of the previous token.
679     *
680     * @return The previous token index.
681     */
682    @Override
683    public int previousIndex() {
684        return tokenPos - 1;
685    }
686
687    /**
688     * Gets the previous token from the String.
689     *
690     * @return The previous sequential token, or null when no more tokens are found.
691     */
692    public String previousToken() {
693        if (hasPrevious()) {
694            return tokens[--tokenPos];
695        }
696        return null;
697    }
698
699    /**
700     * Reads character by character through the String to get the next token.
701     *
702     * @param srcChars  The character array being tokenized.
703     * @param start  The first character of field.
704     * @param len  The length of the character array being tokenized.
705     * @param workArea  A temporary work area.
706     * @param tokenList  The list of parsed tokens.
707     * @return The starting position of the next field (the character
708     *  immediately after the delimiter), or -1 if end of string found.
709     */
710    private int readNextToken(final char[] srcChars, int start, final int len, final StrBuilder workArea, final List<String> tokenList) {
711        // skip all leading whitespace, unless it is the
712        // field delimiter or the quote character
713        while (start < len) {
714            final int removeLen = Math.max(
715                    getIgnoredMatcher().isMatch(srcChars, start, start, len),
716                    getTrimmerMatcher().isMatch(srcChars, start, start, len));
717            if (removeLen == 0 ||
718                getDelimiterMatcher().isMatch(srcChars, start, start, len) > 0 ||
719                getQuoteMatcher().isMatch(srcChars, start, start, len) > 0) {
720                break;
721            }
722            start += removeLen;
723        }
724
725        // handle reaching end
726        if (start >= len) {
727            addToken(tokenList, StringUtils.EMPTY);
728            return -1;
729        }
730
731        // handle empty token
732        final int delimLen = getDelimiterMatcher().isMatch(srcChars, start, start, len);
733        if (delimLen > 0) {
734            addToken(tokenList, StringUtils.EMPTY);
735            return start + delimLen;
736        }
737
738        // handle found token
739        final int quoteLen = getQuoteMatcher().isMatch(srcChars, start, start, len);
740        if (quoteLen > 0) {
741            return readWithQuotes(srcChars, start + quoteLen, len, workArea, tokenList, start, quoteLen);
742        }
743        return readWithQuotes(srcChars, start, len, workArea, tokenList, 0, 0);
744    }
745
746    /**
747     * Reads a possibly quoted string token.
748     *
749     * @param srcChars  The character array being tokenized.
750     * @param start  The first character of field.
751     * @param len  The length of the character array being tokenized.
752     * @param workArea  A temporary work area.
753     * @param tokenList  The list of parsed tokens.
754     * @param quoteStart  The start position of the matched quote, 0 if no quoting.
755     * @param quoteLen  The length of the matched quote, 0 if no quoting.
756     * @return The starting position of the next field (the character
757     *  immediately after the delimiter, or if end of string found,
758     *  then the length of string.
759     */
760    private int readWithQuotes(final char[] srcChars, final int start, final int len, final StrBuilder workArea,
761                               final List<String> tokenList, final int quoteStart, final int quoteLen) {
762        // Loop until we've found the end of the quoted
763        // string or the end of the input
764        workArea.clear();
765        int pos = start;
766        boolean quoting = quoteLen > 0;
767        int trimStart = 0;
768
769        while (pos < len) {
770            // quoting mode can occur several times throughout a string
771            // we must switch between quoting and non-quoting until we
772            // encounter a non-quoted delimiter, or end of string
773            if (quoting) {
774                // In quoting mode
775
776                // If we've found a quote character, see if it's
777                // followed by a second quote.  If so, then we need
778                // to actually put the quote character into the token
779                // rather than end the token.
780                if (isQuote(srcChars, pos, len, quoteStart, quoteLen)) {
781                    if (isQuote(srcChars, pos + quoteLen, len, quoteStart, quoteLen)) {
782                        // matched pair of quotes, thus an escaped quote
783                        workArea.append(srcChars, pos, quoteLen);
784                        pos += quoteLen * 2;
785                        trimStart = workArea.size();
786                        continue;
787                    }
788
789                    // end of quoting
790                    quoting = false;
791                    pos += quoteLen;
792                    continue;
793                }
794
795            } else {
796                // Not in quoting mode
797
798                // check for delimiter, and thus end of token
799                final int delimLen = getDelimiterMatcher().isMatch(srcChars, pos, start, len);
800                if (delimLen > 0) {
801                    // return condition when end of token found
802                    addToken(tokenList, workArea.substring(0, trimStart));
803                    return pos + delimLen;
804                }
805
806                // check for quote, and thus back into quoting mode
807                if (quoteLen > 0 && isQuote(srcChars, pos, len, quoteStart, quoteLen)) {
808                    quoting = true;
809                    pos += quoteLen;
810                    continue;
811                }
812
813                // check for ignored (outside quotes), and ignore
814                final int ignoredLen = getIgnoredMatcher().isMatch(srcChars, pos, start, len);
815                if (ignoredLen > 0) {
816                    pos += ignoredLen;
817                    continue;
818                }
819
820                // check for trimmed character
821                // don't yet know if it's at the end, so copy to workArea
822                // use trimStart to keep track of trim at the end
823                final int trimmedLen = getTrimmerMatcher().isMatch(srcChars, pos, start, len);
824                if (trimmedLen > 0) {
825                    workArea.append(srcChars, pos, trimmedLen);
826                    pos += trimmedLen;
827                    continue;
828                }
829            }
830            // copy regular character from inside quotes
831            workArea.append(srcChars[pos++]);
832            trimStart = workArea.size();
833        }
834
835        // return condition when end of string found
836        addToken(tokenList, workArea.substring(0, trimStart));
837        return -1;
838    }
839
840    /**
841     * Always throws {@link UnsupportedOperationException}.
842     *
843     * @throws UnsupportedOperationException Thrown because this operation is unsupported.
844     */
845    @Override
846    public void remove() {
847        throw new UnsupportedOperationException("remove() is unsupported");
848    }
849
850    /**
851     * Resets this tokenizer, forgetting all parsing and iteration already completed.
852     * <p>
853     * This method allows the same tokenizer to be reused for the same String.
854     * </p>
855     *
856     * @return {@code this} instance.
857     */
858    public StrTokenizer reset() {
859        tokenPos = 0;
860        tokens = null;
861        return this;
862    }
863
864    /**
865     * Reset this tokenizer, giving it a new input string to parse.
866     * In this manner you can re-use a tokenizer with the same settings
867     * on multiple input lines.
868     *
869     * @param input  The new character array to tokenize, not cloned, null sets no text to parse.
870     * @return {@code this} instance.
871     */
872    public StrTokenizer reset(final char[] input) {
873        reset();
874        this.chars = ArrayUtils.clone(input);
875        return this;
876    }
877
878    /**
879     * Reset this tokenizer, giving it a new input string to parse.
880     * In this manner you can re-use a tokenizer with the same settings
881     * on multiple input lines.
882     *
883     * @param input  The new string to tokenize, null sets no text to parse.
884     * @return {@code this} instance.
885     */
886    public StrTokenizer reset(final String input) {
887        reset();
888        if (input != null) {
889            this.chars = input.toCharArray();
890        } else {
891            this.chars = null;
892        }
893        return this;
894    }
895
896    /**
897     * Sets no token and always throws {@link UnsupportedOperationException}.
898     *
899     * @param obj this parameter ignored.
900     * @throws UnsupportedOperationException Thrown because this operation is unsupported.
901     */
902    @Override
903    public void set(final String obj) {
904        throw new UnsupportedOperationException("set() is unsupported");
905    }
906
907    /**
908     * Sets the field delimiter character.
909     *
910     * @param delim  The delimiter character to use.
911     * @return {@code this} instance.
912     */
913    public StrTokenizer setDelimiterChar(final char delim) {
914        return setDelimiterMatcher(StrMatcher.charMatcher(delim));
915    }
916
917    /**
918     * Sets the field delimiter matcher.
919     * <p>
920     * The delimiter is used to separate one token from another.
921     * </p>
922     *
923     * @param delim  The delimiter matcher to use.
924     * @return {@code this} instance.
925     */
926    public StrTokenizer setDelimiterMatcher(final StrMatcher delim) {
927        if (delim == null) {
928            this.delimMatcher = StrMatcher.noneMatcher();
929        } else {
930            this.delimMatcher = delim;
931        }
932        return this;
933    }
934
935    /**
936     * Sets the field delimiter string.
937     *
938     * @param delim  The delimiter string to use.
939     * @return {@code this} instance.
940     */
941    public StrTokenizer setDelimiterString(final String delim) {
942        return setDelimiterMatcher(StrMatcher.stringMatcher(delim));
943    }
944
945    /**
946     * Sets whether the tokenizer should return empty tokens as null.
947     * The default for this property is false.
948     *
949     * @param emptyAsNull  whether empty tokens are returned as null.
950     * @return {@code this} instance.
951     */
952    public StrTokenizer setEmptyTokenAsNull(final boolean emptyAsNull) {
953        this.emptyAsNull = emptyAsNull;
954        return this;
955    }
956
957    /**
958     * Sets the character to ignore.
959     * <p>
960     * This character is ignored when parsing the String, unless it is
961     * within a quoted region.
962     *
963     * @param ignored  The ignored character to use.
964     * @return {@code this} instance.
965     */
966    public StrTokenizer setIgnoredChar(final char ignored) {
967        return setIgnoredMatcher(StrMatcher.charMatcher(ignored));
968    }
969
970    /**
971     * Sets the matcher for characters to ignore.
972     * <p>
973     * These characters are ignored when parsing the String, unless they are
974     * within a quoted region.
975     * </p>
976     *
977     * @param ignored  The ignored matcher to use, null ignored.
978     * @return {@code this} instance.
979     */
980    public StrTokenizer setIgnoredMatcher(final StrMatcher ignored) {
981        if (ignored != null) {
982            this.ignoredMatcher = ignored;
983        }
984        return this;
985    }
986
987    /**
988     * Sets whether the tokenizer should ignore and not return empty tokens.
989     * The default for this property is true.
990     *
991     * @param ignoreEmptyTokens  whether empty tokens are not returned.
992     * @return {@code this} instance.
993     */
994    public StrTokenizer setIgnoreEmptyTokens(final boolean ignoreEmptyTokens) {
995        this.ignoreEmptyTokens = ignoreEmptyTokens;
996        return this;
997    }
998
999    /**
1000     * Sets the quote character to use.
1001     * <p>
1002     * The quote character is used to wrap data between the tokens.
1003     * This enables delimiters to be entered as data.
1004     * </p>
1005     *
1006     * @param quote  The quote character to use.
1007     * @return {@code this} instance.
1008     */
1009    public StrTokenizer setQuoteChar(final char quote) {
1010        return setQuoteMatcher(StrMatcher.charMatcher(quote));
1011    }
1012
1013    /**
1014     * Sets the quote matcher to use.
1015     * <p>
1016     * The quote character is used to wrap data between the tokens.
1017     * This enables delimiters to be entered as data.
1018     * </p>
1019     *
1020     * @param quote  The quote matcher to use, null ignored.
1021     * @return {@code this} instance.
1022     */
1023    public StrTokenizer setQuoteMatcher(final StrMatcher quote) {
1024        if (quote != null) {
1025            this.quoteMatcher = quote;
1026        }
1027        return this;
1028    }
1029
1030    /**
1031     * Sets the matcher for characters to trim.
1032     * <p>
1033     * These characters are trimmed off on each side of the delimiter
1034     * until the token or quote is found.
1035     * </p>
1036     *
1037     * @param trimmer  The trimmer matcher to use, null ignored.
1038     * @return {@code this} instance.
1039     */
1040    public StrTokenizer setTrimmerMatcher(final StrMatcher trimmer) {
1041        if (trimmer != null) {
1042            this.trimmerMatcher = trimmer;
1043        }
1044        return this;
1045    }
1046
1047    /**
1048     * Gets the number of tokens found in the String.
1049     *
1050     * @return The number of matched tokens.
1051     */
1052    public int size() {
1053        checkTokenized();
1054        return tokens.length;
1055    }
1056
1057    /**
1058     * Internal method to performs the tokenization.
1059     * <p>
1060     * Most users of this class do not need to call this method. This method
1061     * will be called automatically by other (public) methods when required.
1062     * </p>
1063     * <p>
1064     * This method exists to allow subclasses to add code before or after the
1065     * tokenization. For example, a subclass could alter the character array,
1066     * offset or count to be parsed, or call the tokenizer multiple times on
1067     * multiple strings. It is also be possible to filter the results.
1068     * </p>
1069     * <p>
1070     * {@link StrTokenizer} will always pass a zero offset and a count
1071     * equal to the length of the array to this method, however a subclass
1072     * may pass other values, or even an entirely different array.
1073     * </p>
1074     *
1075     * @param srcChars  The character array being tokenized, may be null.
1076     * @param offset  The start position within the character array, must be valid.
1077     * @param count  The number of characters to tokenize, must be valid.
1078     * @return The modifiable list of String tokens, unmodifiable if null array or zero count.
1079     */
1080    protected List<String> tokenize(final char[] srcChars, final int offset, final int count) {
1081        if (ArrayUtils.isEmpty(srcChars)) {
1082            return Collections.emptyList();
1083        }
1084        final StrBuilder buf = new StrBuilder();
1085        final List<String> tokenList = new ArrayList<>();
1086        int pos = offset;
1087
1088        // loop around the entire buffer
1089        while (pos >= 0 && pos < count) {
1090            // find next token
1091            pos = readNextToken(srcChars, pos, count, buf, tokenList);
1092
1093            // handle case where end of string is a delimiter
1094            if (pos >= count) {
1095                addToken(tokenList, StringUtils.EMPTY);
1096            }
1097        }
1098        return tokenList;
1099    }
1100
1101    /**
1102     * Gets the String content that the tokenizer is parsing.
1103     *
1104     * @return The string content being parsed.
1105     */
1106    @Override
1107    public String toString() {
1108        if (tokens == null) {
1109            return "StrTokenizer[not tokenized yet]";
1110        }
1111        return "StrTokenizer" + getTokenList();
1112    }
1113
1114}