001/* 002 * Anarres C Preprocessor 003 * Copyright (c) 2007-2015, Shevek 004 * 005 * Licensed under the Apache License, Version 2.0 (the "License"); 006 * you may not use this file except in compliance with the License. 007 * You may obtain a copy of the License at 008 * 009 * http://www.apache.org/licenses/LICENSE-2.0 010 * 011 * Unless required by applicable law or agreed to in writing, software 012 * distributed under the License is distributed on an "AS IS" BASIS, 013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express 014 * or implied. See the License for the specific language governing 015 * permissions and limitations under the License. 016 */ 017package org.anarres.cpp; 018 019import java.io.BufferedReader; 020import java.io.IOException; 021import java.io.Reader; 022import javax.annotation.Nonnull; 023import static org.anarres.cpp.Token.*; 024 025/** Does not handle digraphs. */ 026public class LexerSource extends Source { 027 028 @Nonnull 029 protected static BufferedReader toBufferedReader(@Nonnull Reader r) { 030 if (r instanceof BufferedReader) 031 return (BufferedReader) r; 032 return new BufferedReader(r); 033 } 034 035 private static final boolean DEBUG = false; 036 037 private JoinReader reader; 038 private final boolean ppvalid; 039 private boolean bol; 040 private boolean include; 041 042 private boolean digraphs; 043 044 /* Unread. */ 045 private int u0, u1; 046 private int ucount; 047 048 private int line; 049 private int column; 050 private int lastcolumn; 051 private boolean cr; 052 053 /* ppvalid is: 054 * false in StringLexerSource, 055 * true in FileLexerSource */ 056 public LexerSource(Reader r, boolean ppvalid) { 057 this.reader = new JoinReader(r); 058 this.ppvalid = ppvalid; 059 this.bol = true; 060 this.include = false; 061 062 this.digraphs = true; 063 064 this.ucount = 0; 065 066 this.line = 1; 067 this.column = 0; 068 this.lastcolumn = -1; 069 this.cr = false; 070 } 071 072 @Override 073 /* pp */ void init(Preprocessor pp) { 074 super.init(pp); 075 this.digraphs = pp.getFeature(Feature.DIGRAPHS); 076 this.reader.init(pp, this); 077 } 078 079 /** 080 * Returns the line number of the last read character in this source. 081 * 082 * Lines are numbered from 1. 083 * 084 * @return the line number of the last read character in this source. 085 */ 086 @Override 087 public int getLine() { 088 return line; 089 } 090 091 /** 092 * Returns the column number of the last read character in this source. 093 * 094 * Columns are numbered from 0. 095 * 096 * @return the column number of the last read character in this source. 097 */ 098 @Override 099 public int getColumn() { 100 return column; 101 } 102 103 @Override 104 /* pp */ boolean isNumbered() { 105 return true; 106 } 107 108 /* Error handling. */ 109 private void _error(String msg, boolean error) 110 throws LexerException { 111 int _l = line; 112 int _c = column; 113 if (_c == 0) { 114 _c = lastcolumn; 115 _l--; 116 } else { 117 _c--; 118 } 119 if (error) 120 super.error(_l, _c, msg); 121 else 122 super.warning(_l, _c, msg); 123 } 124 125 /* Allow JoinReader to call this. */ 126 /* pp */ final void error(String msg) 127 throws LexerException { 128 _error(msg, true); 129 } 130 131 /* Allow JoinReader to call this. */ 132 /* pp */ final void warning(String msg) 133 throws LexerException { 134 _error(msg, false); 135 } 136 137 /* A flag for string handling. */ 138 139 /* pp */ void setInclude(boolean b) { 140 this.include = b; 141 } 142 143 /* 144 * private boolean _isLineSeparator(int c) { 145 * return Character.getType(c) == Character.LINE_SEPARATOR 146 * || c == -1; 147 * } 148 */ 149 150 /* XXX Move to JoinReader and canonicalise newlines. */ 151 private static boolean isLineSeparator(int c) { 152 switch ((char) c) { 153 case '\r': 154 case '\n': 155 case '\u2028': 156 case '\u2029': 157 case '\u000B': 158 case '\u000C': 159 case '\u0085': 160 return true; 161 default: 162 return (c == -1); 163 } 164 } 165 166 private int read() 167 throws IOException, 168 LexerException { 169 int c; 170 assert ucount <= 2 : "Illegal ucount: " + ucount; 171 switch (ucount) { 172 case 2: 173 ucount = 1; 174 c = u1; 175 break; 176 case 1: 177 ucount = 0; 178 c = u0; 179 break; 180 default: 181 if (reader == null) 182 c = -1; 183 else 184 c = reader.read(); 185 break; 186 } 187 188 switch (c) { 189 case '\r': 190 cr = true; 191 line++; 192 lastcolumn = column; 193 column = 0; 194 break; 195 case '\n': 196 if (cr) { 197 cr = false; 198 break; 199 } 200 /* fallthrough */ 201 case '\u2028': 202 case '\u2029': 203 case '\u000B': 204 case '\u000C': 205 case '\u0085': 206 cr = false; 207 line++; 208 lastcolumn = column; 209 column = 0; 210 break; 211 case -1: 212 cr = false; 213 break; 214 default: 215 cr = false; 216 column++; 217 break; 218 } 219 220 /* 221 * if (isLineSeparator(c)) { 222 * line++; 223 * lastcolumn = column; 224 * column = 0; 225 * } 226 * else { 227 * column++; 228 * } 229 */ 230 return c; 231 } 232 233 /* You can unget AT MOST one newline. */ 234 private void unread(int c) 235 throws IOException { 236 /* XXX Must unread newlines. */ 237 if (c != -1) { 238 if (isLineSeparator(c)) { 239 line--; 240 column = lastcolumn; 241 cr = false; 242 } else { 243 column--; 244 } 245 switch (ucount) { 246 case 0: 247 u0 = c; 248 ucount = 1; 249 break; 250 case 1: 251 u1 = c; 252 ucount = 2; 253 break; 254 default: 255 throw new IllegalStateException( 256 "Cannot unget another character!" 257 ); 258 } 259 // reader.unread(c); 260 } 261 } 262 263 /* Consumes the rest of the current line into an invalid. */ 264 @Nonnull 265 private Token invalid(StringBuilder text, String reason) 266 throws IOException, 267 LexerException { 268 int d = read(); 269 while (!isLineSeparator(d)) { 270 text.append((char) d); 271 d = read(); 272 } 273 unread(d); 274 return new Token(INVALID, text.toString(), reason); 275 } 276 277 @Nonnull 278 private Token ccomment() 279 throws IOException, 280 LexerException { 281 StringBuilder text = new StringBuilder("/*"); 282 int d; 283 do { 284 do { 285 d = read(); 286 if (d == -1) 287 return new Token(INVALID, text.toString(), 288 "Unterminated comment"); 289 text.append((char) d); 290 } while (d != '*'); 291 do { 292 d = read(); 293 if (d == -1) 294 return new Token(INVALID, text.toString(), 295 "Unterminated comment"); 296 text.append((char) d); 297 } while (d == '*'); 298 } while (d != '/'); 299 return new Token(CCOMMENT, text.toString()); 300 } 301 302 @Nonnull 303 private Token cppcomment() 304 throws IOException, 305 LexerException { 306 StringBuilder text = new StringBuilder("//"); 307 int d = read(); 308 while (!isLineSeparator(d)) { 309 text.append((char) d); 310 d = read(); 311 } 312 unread(d); 313 return new Token(CPPCOMMENT, text.toString()); 314 } 315 316 /** 317 * Lexes an escaped character, appends the lexed escape sequence to 'text' and returns the parsed character value. 318 * 319 * @param text The buffer to which the literal escape sequence is appended. 320 * @return The new parsed character value. 321 * @throws IOException if it goes badly wrong. 322 * @throws LexerException if it goes wrong. 323 */ 324 private int escape(StringBuilder text) 325 throws IOException, 326 LexerException { 327 int d = read(); 328 switch (d) { 329 case 'a': 330 text.append('a'); 331 return 0x07; 332 case 'b': 333 text.append('b'); 334 return '\b'; 335 case 'f': 336 text.append('f'); 337 return '\f'; 338 case 'n': 339 text.append('n'); 340 return '\n'; 341 case 'r': 342 text.append('r'); 343 return '\r'; 344 case 't': 345 text.append('t'); 346 return '\t'; 347 case 'v': 348 text.append('v'); 349 return 0x0b; 350 case '\\': 351 text.append('\\'); 352 return '\\'; 353 354 case '0': 355 case '1': 356 case '2': 357 case '3': 358 case '4': 359 case '5': 360 case '6': 361 case '7': 362 int len = 0; 363 int val = 0; 364 do { 365 val = (val << 3) + Character.digit(d, 8); 366 text.append((char) d); 367 d = read(); 368 } while (++len < 3 && Character.digit(d, 8) != -1); 369 unread(d); 370 return val; 371 372 case 'x': 373 text.append((char) d); 374 len = 0; 375 val = 0; 376 while (len++ < 2) { 377 d = read(); 378 if (Character.digit(d, 16) == -1) { 379 unread(d); 380 break; 381 } 382 val = (val << 4) + Character.digit(d, 16); 383 text.append((char) d); 384 } 385 return val; 386 387 /* Exclude two cases from the warning. */ 388 case '"': 389 text.append('"'); 390 return '"'; 391 case '\'': 392 text.append('\''); 393 return '\''; 394 395 default: 396 warning("Unnecessary escape character " + (char) d); 397 text.append((char) d); 398 return d; 399 } 400 } 401 402 @Nonnull 403 private Token character() 404 throws IOException, 405 LexerException { 406 StringBuilder text = new StringBuilder("'"); 407 int d = read(); 408 if (d == '\\') { 409 text.append('\\'); 410 d = escape(text); 411 } else if (isLineSeparator(d)) { 412 unread(d); 413 return new Token(INVALID, text.toString(), 414 "Unterminated character literal"); 415 } else if (d == '\'') { 416 text.append('\''); 417 return new Token(INVALID, text.toString(), 418 "Empty character literal"); 419 } else if (!Character.isDefined(d)) { 420 text.append('?'); 421 return invalid(text, "Illegal unicode character literal"); 422 } else { 423 text.append((char) d); 424 } 425 426 int e = read(); 427 if (e != '\'') { 428 // error("Illegal character constant"); 429 /* We consume up to the next ' or the rest of the line. */ 430 for (;;) { 431 if (isLineSeparator(e)) { 432 unread(e); 433 break; 434 } 435 text.append((char) e); 436 if (e == '\'') 437 break; 438 e = read(); 439 } 440 return new Token(INVALID, text.toString(), 441 "Illegal character constant " + text); 442 } 443 text.append('\''); 444 /* XXX It this a bad cast? */ 445 return new Token(CHARACTER, 446 text.toString(), Character.valueOf((char) d)); 447 } 448 449 @Nonnull 450 private Token string(char open, char close) 451 throws IOException, 452 LexerException { 453 StringBuilder text = new StringBuilder(); 454 text.append(open); 455 456 StringBuilder buf = new StringBuilder(); 457 458 for (;;) { 459 int c = read(); 460 if (c == close) { 461 break; 462 } else if (c == '\\') { 463 text.append('\\'); 464 if (!include) { 465 char d = (char) escape(text); 466 buf.append(d); 467 } 468 } else if (c == -1) { 469 unread(c); 470 // error("End of file in string literal after " + buf); 471 return new Token(INVALID, text.toString(), 472 "End of file in string literal after " + buf); 473 } else if (isLineSeparator(c)) { 474 unread(c); 475 // error("Unterminated string literal after " + buf); 476 return new Token(INVALID, text.toString(), 477 "Unterminated string literal after " + buf); 478 } else { 479 text.append((char) c); 480 buf.append((char) c); 481 } 482 } 483 text.append(close); 484 switch (close) { 485 case '"': 486 return new Token(STRING, 487 text.toString(), buf.toString()); 488 case '>': 489 return new Token(HEADER, 490 text.toString(), buf.toString()); 491 case '\'': 492 if (buf.length() == 1) 493 return new Token(CHARACTER, 494 text.toString(), buf.toString()); 495 return new Token(SQSTRING, 496 text.toString(), buf.toString()); 497 default: 498 throw new IllegalStateException( 499 "Unknown closing character " + String.valueOf(close)); 500 } 501 } 502 503 @Nonnull 504 private Token _number_suffix(StringBuilder text, NumericValue value, int d) 505 throws IOException, 506 LexerException { 507 int flags = 0; // U, I, L, LL, F, D, MSB 508 for (;;) { 509 if (d == 'U' || d == 'u') { 510 if ((flags & NumericValue.F_UNSIGNED) != 0) 511 warning("Duplicate unsigned suffix " + d); 512 flags |= NumericValue.F_UNSIGNED; 513 text.append((char) d); 514 d = read(); 515 } else if (d == 'L' || d == 'l') { 516 if ((flags & NumericValue.FF_SIZE) != 0) 517 warning("Multiple length suffixes after " + text); 518 text.append((char) d); 519 int e = read(); 520 if (e == d) { // Case must match. Ll is Welsh. 521 flags |= NumericValue.F_LONGLONG; 522 text.append((char) e); 523 d = read(); 524 } else { 525 flags |= NumericValue.F_LONG; 526 d = e; 527 } 528 } else if (d == 'I' || d == 'i') { 529 if ((flags & NumericValue.FF_SIZE) != 0) 530 warning("Multiple length suffixes after " + text); 531 flags |= NumericValue.F_INT; 532 text.append((char) d); 533 d = read(); 534 } else if (d == 'F' || d == 'f') { 535 if ((flags & NumericValue.FF_SIZE) != 0) 536 warning("Multiple length suffixes after " + text); 537 flags |= NumericValue.F_FLOAT; 538 text.append((char) d); 539 d = read(); 540 } else if (d == 'D' || d == 'd') { 541 if ((flags & NumericValue.FF_SIZE) != 0) 542 warning("Multiple length suffixes after " + text); 543 flags |= NumericValue.F_DOUBLE; 544 text.append((char) d); 545 d = read(); 546 } else if (Character.isUnicodeIdentifierPart(d)) { 547 String reason = "Invalid suffix \"" + (char) d + "\" on numeric constant"; 548 // We've encountered something initially identified as a number. 549 // Read in the rest of this token as an identifer but return it as an invalid. 550 while (Character.isUnicodeIdentifierPart(d)) { 551 text.append((char) d); 552 d = read(); 553 } 554 unread(d); 555 return new Token(INVALID, text.toString(), reason); 556 } else { 557 unread(d); 558 value.setFlags(flags); 559 return new Token(NUMBER, 560 text.toString(), value); 561 } 562 } 563 } 564 565 /* Either a decimal part, or a hex exponent. */ 566 @Nonnull 567 private String _number_part(StringBuilder text, int base, boolean sign) 568 throws IOException, 569 LexerException { 570 StringBuilder part = new StringBuilder(); 571 int d = read(); 572 if (sign && (d == '+' || d == '-')) { 573 text.append((char) d); 574 part.append((char) d); 575 d = read(); 576 } 577 while (Character.digit(d, base) != -1) { 578 text.append((char) d); 579 part.append((char) d); 580 d = read(); 581 } 582 unread(d); 583 return part.toString(); 584 } 585 586 /* We do not know whether know the first digit is valid. */ 587 @Nonnull 588 private Token number_hex(char x) 589 throws IOException, 590 LexerException { 591 StringBuilder text = new StringBuilder("0"); 592 text.append(x); 593 String integer = _number_part(text, 16, false); 594 NumericValue value = new NumericValue(16, integer); 595 int d = read(); 596 if (d == '.') { 597 text.append((char) d); 598 String fraction = _number_part(text, 16, false); 599 value.setFractionalPart(fraction); 600 d = read(); 601 } 602 if (d == 'P' || d == 'p') { 603 text.append((char) d); 604 String exponent = _number_part(text, 10, true); 605 value.setExponent(2, exponent); 606 d = read(); 607 } 608 // XXX Make sure it's got enough parts 609 return _number_suffix(text, value, d); 610 } 611 612 private static boolean is_octal(@Nonnull String text) { 613 if (!text.startsWith("0")) 614 return false; 615 for (int i = 0; i < text.length(); i++) 616 if (Character.digit(text.charAt(i), 8) == -1) 617 return false; 618 return true; 619 } 620 621 /* We know we have at least one valid digit, but empty is not 622 * fine. */ 623 @Nonnull 624 private Token number_decimal() 625 throws IOException, 626 LexerException { 627 StringBuilder text = new StringBuilder(); 628 String integer = _number_part(text, 10, false); 629 String fraction = null; 630 String exponent = null; 631 int d = read(); 632 if (d == '.') { 633 text.append((char) d); 634 fraction = _number_part(text, 10, false); 635 d = read(); 636 } 637 if (d == 'E' || d == 'e') { 638 text.append((char) d); 639 exponent = _number_part(text, 10, true); 640 d = read(); 641 } 642 int base = 10; 643 if (fraction == null && exponent == null && integer.startsWith("0")) { 644 if (!is_octal(integer)) 645 warning("Decimal constant starts with 0, but not octal: " + integer); 646 else 647 base = 8; 648 } 649 NumericValue value = new NumericValue(base, integer); 650 if (fraction != null) 651 value.setFractionalPart(fraction); 652 if (exponent != null) 653 value.setExponent(10, exponent); 654 // XXX Make sure it's got enough parts 655 return _number_suffix(text, value, d); 656 } 657 658 /** 659 * Section 6.4.4.1 of C99 660 * 661 * (Not pasted here, but says that the initial negation is a separate token.) 662 * 663 * Section 6.4.4.2 of C99 664 * 665 * A floating constant has a significand part that may be followed 666 * by an exponent part and a suffix that specifies its type. The 667 * components of the significand part may include a digit sequence 668 * representing the whole-number part, followed by a period (.), 669 * followed by a digit sequence representing the fraction part. 670 * 671 * The components of the exponent part are an e, E, p, or P 672 * followed by an exponent consisting of an optionally signed digit 673 * sequence. Either the whole-number part or the fraction part has to 674 * be present; for decimal floating constants, either the period or 675 * the exponent part has to be present. 676 * 677 * The significand part is interpreted as a (decimal or hexadecimal) 678 * rational number; the digit sequence in the exponent part is 679 * interpreted as a decimal integer. For decimal floating constants, 680 * the exponent indicates the power of 10 by which the significand 681 * part is to be scaled. For hexadecimal floating constants, the 682 * exponent indicates the power of 2 by which the significand part is 683 * to be scaled. 684 * 685 * For decimal floating constants, and also for hexadecimal 686 * floating constants when FLT_RADIX is not a power of 2, the result 687 * is either the nearest representable value, or the larger or smaller 688 * representable value immediately adjacent to the nearest representable 689 * value, chosen in an implementation-defined manner. For hexadecimal 690 * floating constants when FLT_RADIX is a power of 2, the result is 691 * correctly rounded. 692 */ 693 @Nonnull 694 private Token number() 695 throws IOException, 696 LexerException { 697 Token tok; 698 int c = read(); 699 if (c == '0') { 700 int d = read(); 701 if (d == 'x' || d == 'X') { 702 tok = number_hex((char) d); 703 } else { 704 unread(d); 705 unread(c); 706 tok = number_decimal(); 707 } 708 } else if (Character.isDigit(c) || c == '.') { 709 unread(c); 710 tok = number_decimal(); 711 } else { 712 throw new LexerException("Asked to parse something as a number which isn't: " + (char) c); 713 } 714 return tok; 715 } 716 717 @Nonnull 718 private Token identifier(int c) 719 throws IOException, 720 LexerException { 721 StringBuilder text = new StringBuilder(); 722 int d; 723 text.append((char) c); 724 for (;;) { 725 d = read(); 726 if (Character.isIdentifierIgnorable(d)) 727 ; else if (Character.isJavaIdentifierPart(d)) 728 text.append((char) d); 729 else 730 break; 731 } 732 unread(d); 733 return new Token(IDENTIFIER, text.toString()); 734 } 735 736 @Nonnull 737 private Token whitespace(int c) 738 throws IOException, 739 LexerException { 740 StringBuilder text = new StringBuilder(); 741 int d; 742 text.append((char) c); 743 for (;;) { 744 d = read(); 745 if (ppvalid && isLineSeparator(d)) /* XXX Ugly. */ 746 break; 747 if (Character.isWhitespace(d)) 748 text.append((char) d); 749 else 750 break; 751 } 752 unread(d); 753 return new Token(WHITESPACE, text.toString()); 754 } 755 756 /* No token processed by cond() contains a newline. */ 757 @Nonnull 758 private Token cond(char c, int yes, int no) 759 throws IOException, 760 LexerException { 761 int d = read(); 762 if (c == d) 763 return new Token(yes); 764 unread(d); 765 return new Token(no); 766 } 767 768 @Override 769 public Token token() 770 throws IOException, 771 LexerException { 772 Token tok = null; 773 774 int _l = line; 775 int _c = column; 776 777 int c = read(); 778 int d; 779 780 switch (c) { 781 case '\n': 782 if (ppvalid) { 783 bol = true; 784 if (include) { 785 tok = new Token(NL, _l, _c, "\n"); 786 } else { 787 int nls = 0; 788 do { 789 nls++; 790 d = read(); 791 } while (d == '\n'); 792 unread(d); 793 char[] text = new char[nls]; 794 for (int i = 0; i < text.length; i++) 795 text[i] = '\n'; 796 // Skip the bol = false below. 797 tok = new Token(NL, _l, _c, new String(text)); 798 } 799 if (DEBUG) 800 System.out.println("lx: Returning NL: " + tok); 801 return tok; 802 } 803 /* Let it be handled as whitespace. */ 804 break; 805 806 case '!': 807 tok = cond('=', NE, '!'); 808 break; 809 810 case '#': 811 if (bol) 812 tok = new Token(HASH); 813 else 814 tok = cond('#', PASTE, '#'); 815 break; 816 817 case '+': 818 d = read(); 819 if (d == '+') 820 tok = new Token(INC); 821 else if (d == '=') 822 tok = new Token(PLUS_EQ); 823 else 824 unread(d); 825 break; 826 case '-': 827 d = read(); 828 if (d == '-') 829 tok = new Token(DEC); 830 else if (d == '=') 831 tok = new Token(SUB_EQ); 832 else if (d == '>') 833 tok = new Token(ARROW); 834 else 835 unread(d); 836 break; 837 838 case '*': 839 tok = cond('=', MULT_EQ, '*'); 840 break; 841 case '/': 842 d = read(); 843 if (d == '*') 844 tok = ccomment(); 845 else if (d == '/') 846 tok = cppcomment(); 847 else if (d == '=') 848 tok = new Token(DIV_EQ); 849 else 850 unread(d); 851 break; 852 853 case '%': 854 d = read(); 855 if (d == '=') 856 tok = new Token(MOD_EQ); 857 else if (digraphs && d == '>') 858 tok = new Token('}'); // digraph 859 else if (digraphs && d == ':') 860 PASTE: 861 { 862 d = read(); 863 if (d != '%') { 864 unread(d); 865 tok = new Token('#'); // digraph 866 break PASTE; 867 } 868 d = read(); 869 if (d != ':') { 870 unread(d); // Unread 2 chars here. 871 unread('%'); 872 tok = new Token('#'); // digraph 873 break PASTE; 874 } 875 tok = new Token(PASTE); // digraph 876 } 877 else 878 unread(d); 879 break; 880 881 case ':': 882 /* :: */ 883 d = read(); 884 if (digraphs && d == '>') 885 tok = new Token(']'); // digraph 886 else 887 unread(d); 888 break; 889 890 case '<': 891 if (include) { 892 tok = string('<', '>'); 893 } else { 894 d = read(); 895 if (d == '=') 896 tok = new Token(LE); 897 else if (d == '<') 898 tok = cond('=', LSH_EQ, LSH); 899 else if (digraphs && d == ':') 900 tok = new Token('['); // digraph 901 else if (digraphs && d == '%') 902 tok = new Token('{'); // digraph 903 else 904 unread(d); 905 } 906 break; 907 908 case '=': 909 tok = cond('=', EQ, '='); 910 break; 911 912 case '>': 913 d = read(); 914 if (d == '=') 915 tok = new Token(GE); 916 else if (d == '>') 917 tok = cond('=', RSH_EQ, RSH); 918 else 919 unread(d); 920 break; 921 922 case '^': 923 tok = cond('=', XOR_EQ, '^'); 924 break; 925 926 case '|': 927 d = read(); 928 if (d == '=') 929 tok = new Token(OR_EQ); 930 else if (d == '|') 931 tok = cond('=', LOR_EQ, LOR); 932 else 933 unread(d); 934 break; 935 case '&': 936 d = read(); 937 if (d == '&') 938 tok = cond('=', LAND_EQ, LAND); 939 else if (d == '=') 940 tok = new Token(AND_EQ); 941 else 942 unread(d); 943 break; 944 945 case '.': 946 d = read(); 947 if (d == '.') 948 tok = cond('.', ELLIPSIS, RANGE); 949 else 950 unread(d); 951 if (Character.isDigit(d)) { 952 unread('.'); 953 tok = number(); 954 } 955 /* XXX decimal fraction */ 956 break; 957 958 case '\'': 959 tok = string('\'', '\''); 960 break; 961 962 case '"': 963 tok = string('"', '"'); 964 break; 965 966 case -1: 967 close(); 968 tok = new Token(EOF, _l, _c, "<eof>"); 969 break; 970 } 971 972 if (tok == null) { 973 if (Character.isWhitespace(c)) { 974 tok = whitespace(c); 975 } else if (Character.isDigit(c)) { 976 unread(c); 977 tok = number(); 978 } else if (Character.isJavaIdentifierStart(c)) { 979 tok = identifier(c); 980 } else { 981 String text = TokenType.getTokenText(c); 982 if (text == null) { 983 if ((c >>> 16) == 0) // Character.isBmpCodePoint() is new in 1.7 984 text = Character.toString((char) c); 985 else 986 text = new String(Character.toChars(c)); 987 } 988 tok = new Token(c, text); 989 } 990 } 991 992 if (bol) { 993 switch (tok.getType()) { 994 case WHITESPACE: 995 case CCOMMENT: 996 break; 997 default: 998 bol = false; 999 break; 1000 } 1001 } 1002 1003 tok.setLocation(_l, _c); 1004 if (DEBUG) 1005 System.out.println("lx: Returning " + tok); 1006 // (new Exception("here")).printStackTrace(System.out); 1007 return tok; 1008 } 1009 1010 @Override 1011 public void close() 1012 throws IOException { 1013 if (reader != null) { 1014 reader.close(); 1015 reader = null; 1016 } 1017 super.close(); 1018 } 1019 1020}