001/*
002 * Anarres C Preprocessor
003 * Copyright (c) 2007-2015, Shevek
004 *
005 * Licensed under the Apache License, Version 2.0 (the "License");
006 * you may not use this file except in compliance with the License.
007 * You may obtain a copy of the License at
008 *
009 *     http://www.apache.org/licenses/LICENSE-2.0
010 *
011 * Unless required by applicable law or agreed to in writing, software
012 * distributed under the License is distributed on an "AS IS" BASIS,
013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express
014 * or implied.  See the License for the specific language governing
015 * permissions and limitations under the License.
016 */
017package org.anarres.cpp;
018
019import java.io.BufferedReader;
020import java.io.IOException;
021import java.io.Reader;
022import javax.annotation.Nonnull;
023import static org.anarres.cpp.Token.*;
024
025/** Does not handle digraphs. */
026public class LexerSource extends Source {
027
028    @Nonnull
029    protected static BufferedReader toBufferedReader(@Nonnull Reader r) {
030        if (r instanceof BufferedReader)
031            return (BufferedReader) r;
032        return new BufferedReader(r);
033    }
034
035    private static final boolean DEBUG = false;
036
037    private JoinReader reader;
038    private final boolean ppvalid;
039    private boolean bol;
040    private boolean include;
041
042    private boolean digraphs;
043
044    /* Unread. */
045    private int u0, u1;
046    private int ucount;
047
048    private int line;
049    private int column;
050    private int lastcolumn;
051    private boolean cr;
052
053    /* ppvalid is:
054     * false in StringLexerSource,
055     * true in FileLexerSource */
056    public LexerSource(Reader r, boolean ppvalid) {
057        this.reader = new JoinReader(r);
058        this.ppvalid = ppvalid;
059        this.bol = true;
060        this.include = false;
061
062        this.digraphs = true;
063
064        this.ucount = 0;
065
066        this.line = 1;
067        this.column = 0;
068        this.lastcolumn = -1;
069        this.cr = false;
070    }
071
072    @Override
073    /* pp */ void init(Preprocessor pp) {
074        super.init(pp);
075        this.digraphs = pp.getFeature(Feature.DIGRAPHS);
076        this.reader.init(pp, this);
077    }
078
079    /**
080     * Returns the line number of the last read character in this source.
081     *
082     * Lines are numbered from 1.
083     *
084     * @return the line number of the last read character in this source.
085     */
086    @Override
087    public int getLine() {
088        return line;
089    }
090
091    /**
092     * Returns the column number of the last read character in this source.
093     *
094     * Columns are numbered from 0.
095     *
096     * @return the column number of the last read character in this source.
097     */
098    @Override
099    public int getColumn() {
100        return column;
101    }
102
103    @Override
104    /* pp */ boolean isNumbered() {
105        return true;
106    }
107
108    /* Error handling. */
109    private void _error(String msg, boolean error)
110            throws LexerException {
111        int _l = line;
112        int _c = column;
113        if (_c == 0) {
114            _c = lastcolumn;
115            _l--;
116        } else {
117            _c--;
118        }
119        if (error)
120            super.error(_l, _c, msg);
121        else
122            super.warning(_l, _c, msg);
123    }
124
125    /* Allow JoinReader to call this. */
126    /* pp */ final void error(String msg)
127            throws LexerException {
128        _error(msg, true);
129    }
130
131    /* Allow JoinReader to call this. */
132    /* pp */ final void warning(String msg)
133            throws LexerException {
134        _error(msg, false);
135    }
136
137    /* A flag for string handling. */
138
139    /* pp */ void setInclude(boolean b) {
140        this.include = b;
141    }
142
143    /*
144     * private boolean _isLineSeparator(int c) {
145     * return Character.getType(c) == Character.LINE_SEPARATOR
146     * || c == -1;
147     * }
148     */
149
150    /* XXX Move to JoinReader and canonicalise newlines. */
151    private static boolean isLineSeparator(int c) {
152        switch ((char) c) {
153            case '\r':
154            case '\n':
155            case '\u2028':
156            case '\u2029':
157            case '\u000B':
158            case '\u000C':
159            case '\u0085':
160                return true;
161            default:
162                return (c == -1);
163        }
164    }
165
166    private int read()
167            throws IOException,
168            LexerException {
169        int c;
170        assert ucount <= 2 : "Illegal ucount: " + ucount;
171        switch (ucount) {
172            case 2:
173                ucount = 1;
174                c = u1;
175                break;
176            case 1:
177                ucount = 0;
178                c = u0;
179                break;
180            default:
181                if (reader == null)
182                    c = -1;
183                else
184                    c = reader.read();
185                break;
186        }
187
188        switch (c) {
189            case '\r':
190                cr = true;
191                line++;
192                lastcolumn = column;
193                column = 0;
194                break;
195            case '\n':
196                if (cr) {
197                    cr = false;
198                    break;
199                }
200            /* fallthrough */
201            case '\u2028':
202            case '\u2029':
203            case '\u000B':
204            case '\u000C':
205            case '\u0085':
206                cr = false;
207                line++;
208                lastcolumn = column;
209                column = 0;
210                break;
211            case -1:
212                cr = false;
213                break;
214            default:
215                cr = false;
216                column++;
217                break;
218        }
219
220        /*
221         * if (isLineSeparator(c)) {
222         * line++;
223         * lastcolumn = column;
224         * column = 0;
225         * }
226         * else {
227         * column++;
228         * }
229         */
230        return c;
231    }
232
233    /* You can unget AT MOST one newline. */
234    private void unread(int c)
235            throws IOException {
236        /* XXX Must unread newlines. */
237        if (c != -1) {
238            if (isLineSeparator(c)) {
239                line--;
240                column = lastcolumn;
241                cr = false;
242            } else {
243                column--;
244            }
245            switch (ucount) {
246                case 0:
247                    u0 = c;
248                    ucount = 1;
249                    break;
250                case 1:
251                    u1 = c;
252                    ucount = 2;
253                    break;
254                default:
255                    throw new IllegalStateException(
256                            "Cannot unget another character!"
257                    );
258            }
259            // reader.unread(c);
260        }
261    }
262
263    /* Consumes the rest of the current line into an invalid. */
264    @Nonnull
265    private Token invalid(StringBuilder text, String reason)
266            throws IOException,
267            LexerException {
268        int d = read();
269        while (!isLineSeparator(d)) {
270            text.append((char) d);
271            d = read();
272        }
273        unread(d);
274        return new Token(INVALID, text.toString(), reason);
275    }
276
277    @Nonnull
278    private Token ccomment()
279            throws IOException,
280            LexerException {
281        StringBuilder text = new StringBuilder("/*");
282        int d;
283        do {
284            do {
285                d = read();
286                if (d == -1)
287                    return new Token(INVALID, text.toString(),
288                            "Unterminated comment");
289                text.append((char) d);
290            } while (d != '*');
291            do {
292                d = read();
293                if (d == -1)
294                    return new Token(INVALID, text.toString(),
295                            "Unterminated comment");
296                text.append((char) d);
297            } while (d == '*');
298        } while (d != '/');
299        return new Token(CCOMMENT, text.toString());
300    }
301
302    @Nonnull
303    private Token cppcomment()
304            throws IOException,
305            LexerException {
306        StringBuilder text = new StringBuilder("//");
307        int d = read();
308        while (!isLineSeparator(d)) {
309            text.append((char) d);
310            d = read();
311        }
312        unread(d);
313        return new Token(CPPCOMMENT, text.toString());
314    }
315
316    /**
317     * Lexes an escaped character, appends the lexed escape sequence to 'text' and returns the parsed character value.
318     *
319     * @param text The buffer to which the literal escape sequence is appended.
320     * @return The new parsed character value.
321     * @throws IOException if it goes badly wrong.
322     * @throws LexerException if it goes wrong.
323     */
324    private int escape(StringBuilder text)
325            throws IOException,
326            LexerException {
327        int d = read();
328        switch (d) {
329            case 'a':
330                text.append('a');
331                return 0x07;
332            case 'b':
333                text.append('b');
334                return '\b';
335            case 'f':
336                text.append('f');
337                return '\f';
338            case 'n':
339                text.append('n');
340                return '\n';
341            case 'r':
342                text.append('r');
343                return '\r';
344            case 't':
345                text.append('t');
346                return '\t';
347            case 'v':
348                text.append('v');
349                return 0x0b;
350            case '\\':
351                text.append('\\');
352                return '\\';
353
354            case '0':
355            case '1':
356            case '2':
357            case '3':
358            case '4':
359            case '5':
360            case '6':
361            case '7':
362                int len = 0;
363                int val = 0;
364                do {
365                    val = (val << 3) + Character.digit(d, 8);
366                    text.append((char) d);
367                    d = read();
368                } while (++len < 3 && Character.digit(d, 8) != -1);
369                unread(d);
370                return val;
371
372            case 'x':
373                text.append((char) d);
374                len = 0;
375                val = 0;
376                while (len++ < 2) {
377                    d = read();
378                    if (Character.digit(d, 16) == -1) {
379                        unread(d);
380                        break;
381                    }
382                    val = (val << 4) + Character.digit(d, 16);
383                    text.append((char) d);
384                }
385                return val;
386
387            /* Exclude two cases from the warning. */
388            case '"':
389                text.append('"');
390                return '"';
391            case '\'':
392                text.append('\'');
393                return '\'';
394
395            default:
396                warning("Unnecessary escape character " + (char) d);
397                text.append((char) d);
398                return d;
399        }
400    }
401
402    @Nonnull
403    private Token character()
404            throws IOException,
405            LexerException {
406        StringBuilder text = new StringBuilder("'");
407        int d = read();
408        if (d == '\\') {
409            text.append('\\');
410            d = escape(text);
411        } else if (isLineSeparator(d)) {
412            unread(d);
413            return new Token(INVALID, text.toString(),
414                    "Unterminated character literal");
415        } else if (d == '\'') {
416            text.append('\'');
417            return new Token(INVALID, text.toString(),
418                    "Empty character literal");
419        } else if (!Character.isDefined(d)) {
420            text.append('?');
421            return invalid(text, "Illegal unicode character literal");
422        } else {
423            text.append((char) d);
424        }
425
426        int e = read();
427        if (e != '\'') {
428            // error("Illegal character constant");
429            /* We consume up to the next ' or the rest of the line. */
430            for (;;) {
431                if (isLineSeparator(e)) {
432                    unread(e);
433                    break;
434                }
435                text.append((char) e);
436                if (e == '\'')
437                    break;
438                e = read();
439            }
440            return new Token(INVALID, text.toString(),
441                    "Illegal character constant " + text);
442        }
443        text.append('\'');
444        /* XXX It this a bad cast? */
445        return new Token(CHARACTER,
446                text.toString(), Character.valueOf((char) d));
447    }
448
449    @Nonnull
450    private Token string(char open, char close)
451            throws IOException,
452            LexerException {
453        StringBuilder text = new StringBuilder();
454        text.append(open);
455
456        StringBuilder buf = new StringBuilder();
457
458        for (;;) {
459            int c = read();
460            if (c == close) {
461                break;
462            } else if (c == '\\') {
463                text.append('\\');
464                if (!include) {
465                    char d = (char) escape(text);
466                    buf.append(d);
467                }
468            } else if (c == -1) {
469                unread(c);
470                // error("End of file in string literal after " + buf);
471                return new Token(INVALID, text.toString(),
472                        "End of file in string literal after " + buf);
473            } else if (isLineSeparator(c)) {
474                unread(c);
475                // error("Unterminated string literal after " + buf);
476                return new Token(INVALID, text.toString(),
477                        "Unterminated string literal after " + buf);
478            } else {
479                text.append((char) c);
480                buf.append((char) c);
481            }
482        }
483        text.append(close);
484        switch (close) {
485            case '"':
486                return new Token(STRING,
487                        text.toString(), buf.toString());
488            case '>':
489                return new Token(HEADER,
490                        text.toString(), buf.toString());
491            case '\'':
492                if (buf.length() == 1)
493                    return new Token(CHARACTER,
494                            text.toString(), buf.toString());
495                return new Token(SQSTRING,
496                        text.toString(), buf.toString());
497            default:
498                throw new IllegalStateException(
499                        "Unknown closing character " + String.valueOf(close));
500        }
501    }
502
503    @Nonnull
504    private Token _number_suffix(StringBuilder text, NumericValue value, int d)
505            throws IOException,
506            LexerException {
507        int flags = 0;  // U, I, L, LL, F, D, MSB
508        for (;;) {
509            if (d == 'U' || d == 'u') {
510                if ((flags & NumericValue.F_UNSIGNED) != 0)
511                    warning("Duplicate unsigned suffix " + d);
512                flags |= NumericValue.F_UNSIGNED;
513                text.append((char) d);
514                d = read();
515            } else if (d == 'L' || d == 'l') {
516                if ((flags & NumericValue.FF_SIZE) != 0)
517                    warning("Multiple length suffixes after " + text);
518                text.append((char) d);
519                int e = read();
520                if (e == d) {   // Case must match. Ll is Welsh.
521                    flags |= NumericValue.F_LONGLONG;
522                    text.append((char) e);
523                    d = read();
524                } else {
525                    flags |= NumericValue.F_LONG;
526                    d = e;
527                }
528            } else if (d == 'I' || d == 'i') {
529                if ((flags & NumericValue.FF_SIZE) != 0)
530                    warning("Multiple length suffixes after " + text);
531                flags |= NumericValue.F_INT;
532                text.append((char) d);
533                d = read();
534            } else if (d == 'F' || d == 'f') {
535                if ((flags & NumericValue.FF_SIZE) != 0)
536                    warning("Multiple length suffixes after " + text);
537                flags |= NumericValue.F_FLOAT;
538                text.append((char) d);
539                d = read();
540            } else if (d == 'D' || d == 'd') {
541                if ((flags & NumericValue.FF_SIZE) != 0)
542                    warning("Multiple length suffixes after " + text);
543                flags |= NumericValue.F_DOUBLE;
544                text.append((char) d);
545                d = read();
546            } else if (Character.isUnicodeIdentifierPart(d)) {
547                String reason = "Invalid suffix \"" + (char) d + "\" on numeric constant";
548                // We've encountered something initially identified as a number.
549                // Read in the rest of this token as an identifer but return it as an invalid.
550                while (Character.isUnicodeIdentifierPart(d)) {
551                    text.append((char) d);
552                    d = read();
553                }
554                unread(d);
555                return new Token(INVALID, text.toString(), reason);
556            } else {
557                unread(d);
558                value.setFlags(flags);
559                return new Token(NUMBER,
560                        text.toString(), value);
561            }
562        }
563    }
564
565    /* Either a decimal part, or a hex exponent. */
566    @Nonnull
567    private String _number_part(StringBuilder text, int base, boolean sign)
568            throws IOException,
569            LexerException {
570        StringBuilder part = new StringBuilder();
571        int d = read();
572        if (sign && (d == '+' || d == '-')) {
573            text.append((char) d);
574            part.append((char) d);
575            d = read();
576        }
577        while (Character.digit(d, base) != -1) {
578            text.append((char) d);
579            part.append((char) d);
580            d = read();
581        }
582        unread(d);
583        return part.toString();
584    }
585
586    /* We do not know whether know the first digit is valid. */
587    @Nonnull
588    private Token number_hex(char x)
589            throws IOException,
590            LexerException {
591        StringBuilder text = new StringBuilder("0");
592        text.append(x);
593        String integer = _number_part(text, 16, false);
594        NumericValue value = new NumericValue(16, integer);
595        int d = read();
596        if (d == '.') {
597            text.append((char) d);
598            String fraction = _number_part(text, 16, false);
599            value.setFractionalPart(fraction);
600            d = read();
601        }
602        if (d == 'P' || d == 'p') {
603            text.append((char) d);
604            String exponent = _number_part(text, 10, true);
605            value.setExponent(2, exponent);
606            d = read();
607        }
608        // XXX Make sure it's got enough parts
609        return _number_suffix(text, value, d);
610    }
611
612    private static boolean is_octal(@Nonnull String text) {
613        if (!text.startsWith("0"))
614            return false;
615        for (int i = 0; i < text.length(); i++)
616            if (Character.digit(text.charAt(i), 8) == -1)
617                return false;
618        return true;
619    }
620
621    /* We know we have at least one valid digit, but empty is not
622     * fine. */
623    @Nonnull
624    private Token number_decimal()
625            throws IOException,
626            LexerException {
627        StringBuilder text = new StringBuilder();
628        String integer = _number_part(text, 10, false);
629        String fraction = null;
630        String exponent = null;
631        int d = read();
632        if (d == '.') {
633            text.append((char) d);
634            fraction = _number_part(text, 10, false);
635            d = read();
636        }
637        if (d == 'E' || d == 'e') {
638            text.append((char) d);
639            exponent = _number_part(text, 10, true);
640            d = read();
641        }
642        int base = 10;
643        if (fraction == null && exponent == null && integer.startsWith("0")) {
644            if (!is_octal(integer))
645                warning("Decimal constant starts with 0, but not octal: " + integer);
646            else
647                base = 8;
648        }
649        NumericValue value = new NumericValue(base, integer);
650        if (fraction != null)
651            value.setFractionalPart(fraction);
652        if (exponent != null)
653            value.setExponent(10, exponent);
654        // XXX Make sure it's got enough parts
655        return _number_suffix(text, value, d);
656    }
657
658    /**
659     * Section 6.4.4.1 of C99
660     *
661     * (Not pasted here, but says that the initial negation is a separate token.)
662     *
663     * Section 6.4.4.2 of C99
664     *
665     * A floating constant has a significand part that may be followed
666     * by an exponent part and a suffix that specifies its type. The
667     * components of the significand part may include a digit sequence
668     * representing the whole-number part, followed by a period (.),
669     * followed by a digit sequence representing the fraction part.
670     *
671     * The components of the exponent part are an e, E, p, or P
672     * followed by an exponent consisting of an optionally signed digit
673     * sequence. Either the whole-number part or the fraction part has to
674     * be present; for decimal floating constants, either the period or
675     * the exponent part has to be present.
676     *
677     * The significand part is interpreted as a (decimal or hexadecimal)
678     * rational number; the digit sequence in the exponent part is
679     * interpreted as a decimal integer. For decimal floating constants,
680     * the exponent indicates the power of 10 by which the significand
681     * part is to be scaled. For hexadecimal floating constants, the
682     * exponent indicates the power of 2 by which the significand part is
683     * to be scaled.
684     *
685     * For decimal floating constants, and also for hexadecimal
686     * floating constants when FLT_RADIX is not a power of 2, the result
687     * is either the nearest representable value, or the larger or smaller
688     * representable value immediately adjacent to the nearest representable
689     * value, chosen in an implementation-defined manner. For hexadecimal
690     * floating constants when FLT_RADIX is a power of 2, the result is
691     * correctly rounded.
692     */
693    @Nonnull
694    private Token number()
695            throws IOException,
696            LexerException {
697        Token tok;
698        int c = read();
699        if (c == '0') {
700            int d = read();
701            if (d == 'x' || d == 'X') {
702                tok = number_hex((char) d);
703            } else {
704                unread(d);
705                unread(c);
706                tok = number_decimal();
707            }
708        } else if (Character.isDigit(c) || c == '.') {
709            unread(c);
710            tok = number_decimal();
711        } else {
712            throw new LexerException("Asked to parse something as a number which isn't: " + (char) c);
713        }
714        return tok;
715    }
716
717    @Nonnull
718    private Token identifier(int c)
719            throws IOException,
720            LexerException {
721        StringBuilder text = new StringBuilder();
722        int d;
723        text.append((char) c);
724        for (;;) {
725            d = read();
726            if (Character.isIdentifierIgnorable(d))
727                                ; else if (Character.isJavaIdentifierPart(d))
728                text.append((char) d);
729            else
730                break;
731        }
732        unread(d);
733        return new Token(IDENTIFIER, text.toString());
734    }
735
736    @Nonnull
737    private Token whitespace(int c)
738            throws IOException,
739            LexerException {
740        StringBuilder text = new StringBuilder();
741        int d;
742        text.append((char) c);
743        for (;;) {
744            d = read();
745            if (ppvalid && isLineSeparator(d)) /* XXX Ugly. */
746                break;
747            if (Character.isWhitespace(d))
748                text.append((char) d);
749            else
750                break;
751        }
752        unread(d);
753        return new Token(WHITESPACE, text.toString());
754    }
755
756    /* No token processed by cond() contains a newline. */
757    @Nonnull
758    private Token cond(char c, int yes, int no)
759            throws IOException,
760            LexerException {
761        int d = read();
762        if (c == d)
763            return new Token(yes);
764        unread(d);
765        return new Token(no);
766    }
767
768    @Override
769    public Token token()
770            throws IOException,
771            LexerException {
772        Token tok = null;
773
774        int _l = line;
775        int _c = column;
776
777        int c = read();
778        int d;
779
780        switch (c) {
781            case '\n':
782                if (ppvalid) {
783                    bol = true;
784                    if (include) {
785                        tok = new Token(NL, _l, _c, "\n");
786                    } else {
787                        int nls = 0;
788                        do {
789                            nls++;
790                            d = read();
791                        } while (d == '\n');
792                        unread(d);
793                        char[] text = new char[nls];
794                        for (int i = 0; i < text.length; i++)
795                            text[i] = '\n';
796                        // Skip the bol = false below.
797                        tok = new Token(NL, _l, _c, new String(text));
798                    }
799                    if (DEBUG)
800                        System.out.println("lx: Returning NL: " + tok);
801                    return tok;
802                }
803                /* Let it be handled as whitespace. */
804                break;
805
806            case '!':
807                tok = cond('=', NE, '!');
808                break;
809
810            case '#':
811                if (bol)
812                    tok = new Token(HASH);
813                else
814                    tok = cond('#', PASTE, '#');
815                break;
816
817            case '+':
818                d = read();
819                if (d == '+')
820                    tok = new Token(INC);
821                else if (d == '=')
822                    tok = new Token(PLUS_EQ);
823                else
824                    unread(d);
825                break;
826            case '-':
827                d = read();
828                if (d == '-')
829                    tok = new Token(DEC);
830                else if (d == '=')
831                    tok = new Token(SUB_EQ);
832                else if (d == '>')
833                    tok = new Token(ARROW);
834                else
835                    unread(d);
836                break;
837
838            case '*':
839                tok = cond('=', MULT_EQ, '*');
840                break;
841            case '/':
842                d = read();
843                if (d == '*')
844                    tok = ccomment();
845                else if (d == '/')
846                    tok = cppcomment();
847                else if (d == '=')
848                    tok = new Token(DIV_EQ);
849                else
850                    unread(d);
851                break;
852
853            case '%':
854                d = read();
855                if (d == '=')
856                    tok = new Token(MOD_EQ);
857                else if (digraphs && d == '>')
858                    tok = new Token('}');       // digraph
859                else if (digraphs && d == ':')
860                    PASTE:
861                    {
862                        d = read();
863                        if (d != '%') {
864                            unread(d);
865                            tok = new Token('#');       // digraph
866                            break PASTE;
867                        }
868                        d = read();
869                        if (d != ':') {
870                            unread(d);  // Unread 2 chars here.
871                            unread('%');
872                            tok = new Token('#');       // digraph
873                            break PASTE;
874                        }
875                        tok = new Token(PASTE); // digraph
876                    }
877                else
878                    unread(d);
879                break;
880
881            case ':':
882                /* :: */
883                d = read();
884                if (digraphs && d == '>')
885                    tok = new Token(']');       // digraph
886                else
887                    unread(d);
888                break;
889
890            case '<':
891                if (include) {
892                    tok = string('<', '>');
893                } else {
894                    d = read();
895                    if (d == '=')
896                        tok = new Token(LE);
897                    else if (d == '<')
898                        tok = cond('=', LSH_EQ, LSH);
899                    else if (digraphs && d == ':')
900                        tok = new Token('[');   // digraph
901                    else if (digraphs && d == '%')
902                        tok = new Token('{');   // digraph
903                    else
904                        unread(d);
905                }
906                break;
907
908            case '=':
909                tok = cond('=', EQ, '=');
910                break;
911
912            case '>':
913                d = read();
914                if (d == '=')
915                    tok = new Token(GE);
916                else if (d == '>')
917                    tok = cond('=', RSH_EQ, RSH);
918                else
919                    unread(d);
920                break;
921
922            case '^':
923                tok = cond('=', XOR_EQ, '^');
924                break;
925
926            case '|':
927                d = read();
928                if (d == '=')
929                    tok = new Token(OR_EQ);
930                else if (d == '|')
931                    tok = cond('=', LOR_EQ, LOR);
932                else
933                    unread(d);
934                break;
935            case '&':
936                d = read();
937                if (d == '&')
938                    tok = cond('=', LAND_EQ, LAND);
939                else if (d == '=')
940                    tok = new Token(AND_EQ);
941                else
942                    unread(d);
943                break;
944
945            case '.':
946                d = read();
947                if (d == '.')
948                    tok = cond('.', ELLIPSIS, RANGE);
949                else
950                    unread(d);
951                if (Character.isDigit(d)) {
952                    unread('.');
953                    tok = number();
954                }
955                /* XXX decimal fraction */
956                break;
957
958            case '\'':
959                tok = string('\'', '\'');
960                break;
961
962            case '"':
963                tok = string('"', '"');
964                break;
965
966            case -1:
967                close();
968                tok = new Token(EOF, _l, _c, "<eof>");
969                break;
970        }
971
972        if (tok == null) {
973            if (Character.isWhitespace(c)) {
974                tok = whitespace(c);
975            } else if (Character.isDigit(c)) {
976                unread(c);
977                tok = number();
978            } else if (Character.isJavaIdentifierStart(c)) {
979                tok = identifier(c);
980            } else {
981                String text = TokenType.getTokenText(c);
982                if (text == null) {
983                    if ((c >>> 16) == 0)    // Character.isBmpCodePoint() is new in 1.7
984                        text = Character.toString((char) c);
985                    else
986                        text = new String(Character.toChars(c));
987                }
988                tok = new Token(c, text);
989            }
990        }
991
992        if (bol) {
993            switch (tok.getType()) {
994                case WHITESPACE:
995                case CCOMMENT:
996                    break;
997                default:
998                    bol = false;
999                    break;
1000            }
1001        }
1002
1003        tok.setLocation(_l, _c);
1004        if (DEBUG)
1005            System.out.println("lx: Returning " + tok);
1006        // (new Exception("here")).printStackTrace(System.out);
1007        return tok;
1008    }
1009
1010    @Override
1011    public void close()
1012            throws IOException {
1013        if (reader != null) {
1014            reader.close();
1015            reader = null;
1016        }
1017        super.close();
1018    }
1019
1020}