001/* 002 * Anarres C Preprocessor 003 * Copyright (c) 2007-2015, Shevek 004 * 005 * Licensed under the Apache License, Version 2.0 (the "License"); 006 * you may not use this file except in compliance with the License. 007 * You may obtain a copy of the License at 008 * 009 * http://www.apache.org/licenses/LICENSE-2.0 010 * 011 * Unless required by applicable law or agreed to in writing, software 012 * distributed under the License is distributed on an "AS IS" BASIS, 013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express 014 * or implied. See the License for the specific language governing 015 * permissions and limitations under the License. 016 */ 017package org.anarres.cpp; 018 019import java.io.Closeable; 020import java.io.File; 021import java.io.IOException; 022import java.util.ArrayList; 023import java.util.Arrays; 024import java.util.Collection; 025import java.util.Collections; 026import java.util.EnumSet; 027import java.util.HashMap; 028import java.util.HashSet; 029import java.util.List; 030import java.util.Map; 031import java.util.Set; 032import java.util.Stack; 033import java.util.TreeMap; 034import javax.annotation.CheckForNull; 035import javax.annotation.Nonnull; 036import org.slf4j.Logger; 037import org.slf4j.LoggerFactory; 038import static org.anarres.cpp.PreprocessorCommand.*; 039import org.anarres.cpp.PreprocessorListener.SourceChangeEvent; 040import static org.anarres.cpp.Token.*; 041 042/** 043 * A C Preprocessor. 044 * The Preprocessor outputs a token stream which does not need 045 * re-lexing for C or C++. Alternatively, the output text may be 046 * reconstructed by concatenating the {@link Token#getText() text} 047 * values of the returned {@link Token Tokens}. (See 048 * {@link CppReader}, which does this.) 049 */ 050/* 051 * Source file name and line number information is conveyed by lines of the form 052 * 053 * # linenum filename flags 054 * 055 * These are called linemarkers. They are inserted as needed into 056 * the output (but never within a string or character constant). They 057 * mean that the following line originated in file filename at line 058 * linenum. filename will never contain any non-printing characters; 059 * they are replaced with octal escape sequences. 060 * 061 * After the file name comes zero or more flags, which are `1', `2', 062 * `3', or `4'. If there are multiple flags, spaces separate them. Here 063 * is what the flags mean: 064 * 065 * `1' 066 * This indicates the start of a new file. 067 * `2' 068 * This indicates returning to a file (after having included another 069 * file). 070 * `3' 071 * This indicates that the following text comes from a system header 072 * file, so certain warnings should be suppressed. 073 * `4' 074 * This indicates that the following text should be treated as being 075 * wrapped in an implicit extern "C" block. 076 */ 077public class Preprocessor implements Closeable { 078 079 private static final Logger LOG = LoggerFactory.getLogger(Preprocessor.class); 080 081 private static final Source INTERNAL = new Source() { 082 @Override 083 public Token token() 084 throws IOException, 085 LexerException { 086 throw new LexerException("Cannot read from " + getName()); 087 } 088 089 @Override 090 public String getPath() { 091 return "<internal-data>"; 092 } 093 094 @Override 095 public String getName() { 096 return "internal data"; 097 } 098 }; 099 private static final Macro __LINE__ = new Macro(INTERNAL, "__LINE__"); 100 private static final Macro __FILE__ = new Macro(INTERNAL, "__FILE__"); 101 private static final Macro __COUNTER__ = new Macro(INTERNAL, "__COUNTER__"); 102 103 private final List<Source> inputs; 104 105 /* The fundamental engine. */ 106 private final Map<String, Macro> macros; 107 private final Stack<State> states; 108 private Source source; 109 110 /* Miscellaneous support. */ 111 private int counter; 112 private final Set<String> onceseenpaths = new HashSet<String>(); 113 private final List<VirtualFile> includes = new ArrayList<VirtualFile>(); 114 115 /* Support junk to make it work like cpp */ 116 private List<String> quoteincludepath; /* -iquote */ 117 118 private List<String> sysincludepath; /* -I */ 119 120 private List<String> frameworkspath; 121 private final Set<Feature> features; 122 private final Set<Warning> warnings; 123 private VirtualFileSystem filesystem; 124 private PreprocessorListener listener; 125 126 public Preprocessor() { 127 this.inputs = new ArrayList<Source>(); 128 129 this.macros = new HashMap<String, Macro>(); 130 macros.put(__LINE__.getName(), __LINE__); 131 macros.put(__FILE__.getName(), __FILE__); 132 macros.put(__COUNTER__.getName(), __COUNTER__); 133 this.states = new Stack<State>(); 134 states.push(new State()); 135 this.source = null; 136 137 this.counter = 0; 138 139 this.quoteincludepath = new ArrayList<String>(); 140 this.sysincludepath = new ArrayList<String>(); 141 this.frameworkspath = new ArrayList<String>(); 142 this.features = EnumSet.noneOf(Feature.class); 143 this.warnings = EnumSet.noneOf(Warning.class); 144 this.filesystem = new JavaFileSystem(); 145 this.listener = null; 146 } 147 148 public Preprocessor(@Nonnull Source initial) { 149 this(); 150 addInput(initial); 151 } 152 153 /** Equivalent to 154 * 'new Preprocessor(new {@link FileLexerSource}(file))' 155 */ 156 public Preprocessor(@Nonnull File file) 157 throws IOException { 158 this(new FileLexerSource(file)); 159 } 160 161 /** 162 * Sets the VirtualFileSystem used by this Preprocessor. 163 */ 164 public void setFileSystem(@Nonnull VirtualFileSystem filesystem) { 165 this.filesystem = filesystem; 166 } 167 168 /** 169 * Returns the VirtualFileSystem used by this Preprocessor. 170 */ 171 @Nonnull 172 public VirtualFileSystem getFileSystem() { 173 return filesystem; 174 } 175 176 /** 177 * Sets the PreprocessorListener which handles events for 178 * this Preprocessor. 179 * 180 * The listener is notified of warnings, errors and source 181 * changes, amongst other things. 182 */ 183 public void setListener(@Nonnull PreprocessorListener listener) { 184 this.listener = listener; 185 Source s = source; 186 while (s != null) { 187 // s.setListener(listener); 188 s.init(this); 189 s = s.getParent(); 190 } 191 } 192 193 /** 194 * Returns the PreprocessorListener which handles events for 195 * this Preprocessor. 196 */ 197 @Nonnull 198 public PreprocessorListener getListener() { 199 return listener; 200 } 201 202 /** 203 * Returns the feature-set for this Preprocessor. 204 * 205 * This set may be freely modified by user code. 206 */ 207 @Nonnull 208 public Set<Feature> getFeatures() { 209 return features; 210 } 211 212 /** 213 * Adds a feature to the feature-set of this Preprocessor. 214 */ 215 public void addFeature(@Nonnull Feature f) { 216 features.add(f); 217 } 218 219 /** 220 * Adds features to the feature-set of this Preprocessor. 221 */ 222 public void addFeatures(@Nonnull Collection<Feature> f) { 223 features.addAll(f); 224 } 225 226 /** 227 * Adds features to the feature-set of this Preprocessor. 228 */ 229 public void addFeatures(Feature... f) { 230 addFeatures(Arrays.asList(f)); 231 } 232 233 /** 234 * Returns true if the given feature is in 235 * the feature-set of this Preprocessor. 236 */ 237 public boolean getFeature(@Nonnull Feature f) { 238 return features.contains(f); 239 } 240 241 /** 242 * Returns the warning-set for this Preprocessor. 243 * 244 * This set may be freely modified by user code. 245 */ 246 @Nonnull 247 public Set<Warning> getWarnings() { 248 return warnings; 249 } 250 251 /** 252 * Adds a warning to the warning-set of this Preprocessor. 253 */ 254 public void addWarning(@Nonnull Warning w) { 255 warnings.add(w); 256 } 257 258 /** 259 * Adds warnings to the warning-set of this Preprocessor. 260 */ 261 public void addWarnings(@Nonnull Collection<Warning> w) { 262 warnings.addAll(w); 263 } 264 265 /** 266 * Returns true if the given warning is in 267 * the warning-set of this Preprocessor. 268 */ 269 public boolean getWarning(@Nonnull Warning w) { 270 return warnings.contains(w); 271 } 272 273 /** 274 * Adds input for the Preprocessor. 275 * 276 * Inputs are processed in the order in which they are added. 277 */ 278 public void addInput(@Nonnull Source source) { 279 source.init(this); 280 inputs.add(source); 281 } 282 283 /** 284 * Adds input for the Preprocessor. 285 * 286 * @see #addInput(Source) 287 */ 288 public void addInput(@Nonnull File file) 289 throws IOException { 290 addInput(new FileLexerSource(file)); 291 } 292 293 /** 294 * Handles an error. 295 * 296 * If a PreprocessorListener is installed, it receives the 297 * error. Otherwise, an exception is thrown. 298 */ 299 protected void error(int line, int column, @Nonnull String msg) 300 throws LexerException { 301 if (listener != null) 302 listener.handleError(source, line, column, msg); 303 else 304 throw new LexerException("Error at " + line + ":" + column + ": " + msg); 305 } 306 307 /** 308 * Handles an error. 309 * 310 * If a PreprocessorListener is installed, it receives the 311 * error. Otherwise, an exception is thrown. 312 * 313 * @see #error(int, int, String) 314 */ 315 protected void error(@Nonnull Token tok, @Nonnull String msg) 316 throws LexerException { 317 error(tok.getLine(), tok.getColumn(), msg); 318 } 319 320 /** 321 * Handles a warning. 322 * 323 * If a PreprocessorListener is installed, it receives the 324 * warning. Otherwise, an exception is thrown. 325 */ 326 protected void warning(int line, int column, @Nonnull String msg) 327 throws LexerException { 328 if (warnings.contains(Warning.ERROR)) 329 error(line, column, msg); 330 else if (listener != null) 331 listener.handleWarning(source, line, column, msg); 332 else 333 throw new LexerException("Warning at " + line + ":" + column + ": " + msg); 334 } 335 336 /** 337 * Handles a warning. 338 * 339 * If a PreprocessorListener is installed, it receives the 340 * warning. Otherwise, an exception is thrown. 341 * 342 * @see #warning(int, int, String) 343 */ 344 protected void warning(@Nonnull Token tok, @Nonnull String msg) 345 throws LexerException { 346 warning(tok.getLine(), tok.getColumn(), msg); 347 } 348 349 /** 350 * Adds a Macro to this Preprocessor. 351 * 352 * The given {@link Macro} object encapsulates both the name 353 * and the expansion. 354 * 355 * @throws LexerException if the definition fails or is otherwise illegal. 356 */ 357 public void addMacro(@Nonnull Macro m) throws LexerException { 358 // System.out.println("Macro " + m); 359 String name = m.getName(); 360 /* Already handled as a source error in macro(). */ 361 if ("defined".equals(name)) 362 throw new LexerException("Cannot redefine name 'defined'"); 363 macros.put(m.getName(), m); 364 } 365 366 /** 367 * Defines the given name as a macro. 368 * 369 * The String value is lexed into a token stream, which is 370 * used as the macro expansion. 371 * 372 * @throws LexerException if the definition fails or is otherwise illegal. 373 */ 374 public void addMacro(@Nonnull String name, @Nonnull String value) 375 throws LexerException { 376 try { 377 Macro m = new Macro(name); 378 StringLexerSource s = new StringLexerSource(value); 379 for (;;) { 380 Token tok = s.token(); 381 if (tok.getType() == EOF) 382 break; 383 m.addToken(tok); 384 } 385 addMacro(m); 386 } catch (IOException e) { 387 throw new LexerException(e); 388 } 389 } 390 391 /** 392 * Defines the given name as a macro, with the value <code>1</code>. 393 * 394 * This is a convnience method, and is equivalent to 395 * <code>addMacro(name, "1")</code>. 396 * 397 * @throws LexerException if the definition fails or is otherwise illegal. 398 */ 399 public void addMacro(@Nonnull String name) 400 throws LexerException { 401 addMacro(name, "1"); 402 } 403 404 /** 405 * Sets the user include path used by this Preprocessor. 406 */ 407 /* Note for future: Create an IncludeHandler? */ 408 public void setQuoteIncludePath(@Nonnull List<String> path) { 409 this.quoteincludepath = path; 410 } 411 412 /** 413 * Returns the user include-path of this Preprocessor. 414 * 415 * This list may be freely modified by user code. 416 */ 417 @Nonnull 418 public List<String> getQuoteIncludePath() { 419 return quoteincludepath; 420 } 421 422 /** 423 * Sets the system include path used by this Preprocessor. 424 */ 425 /* Note for future: Create an IncludeHandler? */ 426 public void setSystemIncludePath(@Nonnull List<String> path) { 427 this.sysincludepath = path; 428 } 429 430 /** 431 * Returns the system include-path of this Preprocessor. 432 * 433 * This list may be freely modified by user code. 434 */ 435 @Nonnull 436 public List<String> getSystemIncludePath() { 437 return sysincludepath; 438 } 439 440 /** 441 * Sets the Objective-C frameworks path used by this Preprocessor. 442 */ 443 /* Note for future: Create an IncludeHandler? */ 444 public void setFrameworksPath(@Nonnull List<String> path) { 445 this.frameworkspath = path; 446 } 447 448 /** 449 * Returns the Objective-C frameworks path used by this 450 * Preprocessor. 451 * 452 * This list may be freely modified by user code. 453 */ 454 @Nonnull 455 public List<String> getFrameworksPath() { 456 return frameworkspath; 457 } 458 459 /** 460 * Returns the Map of Macros parsed during the run of this 461 * Preprocessor. 462 * 463 * @return The {@link Map} of macros currently defined. 464 */ 465 @Nonnull 466 public Map<String, Macro> getMacros() { 467 return macros; 468 } 469 470 /** 471 * Returns the named macro. 472 * 473 * While you can modify the returned object, unexpected things 474 * might happen if you do. 475 * 476 * @return the Macro object, or null if not found. 477 */ 478 @CheckForNull 479 public Macro getMacro(@Nonnull String name) { 480 return macros.get(name); 481 } 482 483 /** 484 * Returns the list of {@link VirtualFile VirtualFiles} which have been 485 * included by this Preprocessor. 486 * 487 * This does not include any {@link Source} provided to the constructor 488 * or {@link #addInput(java.io.File)} or {@link #addInput(Source)}. 489 */ 490 @Nonnull 491 public List<? extends VirtualFile> getIncludes() { 492 return includes; 493 } 494 495 /* States */ 496 private void push_state() { 497 State top = states.peek(); 498 states.push(new State(top)); 499 } 500 501 private void pop_state() 502 throws LexerException { 503 State s = states.pop(); 504 if (states.isEmpty()) { 505 error(0, 0, "#" + "endif without #" + "if"); 506 states.push(s); 507 } 508 } 509 510 private boolean isActive() { 511 State state = states.peek(); 512 return state.isParentActive() && state.isActive(); 513 } 514 515 516 /* Sources */ 517 /** 518 * Returns the top Source on the input stack. 519 * 520 * @see Source 521 * @see #push_source(Source,boolean) 522 * @see #pop_source() 523 * 524 * @return the top Source on the input stack. 525 */ 526 // @CheckForNull 527 protected Source getSource() { 528 return source; 529 } 530 531 /** 532 * Pushes a Source onto the input stack. 533 * 534 * @param source the new Source to push onto the top of the input stack. 535 * @param autopop if true, the Source is automatically removed from the input stack at EOF. 536 * @see #getSource() 537 * @see #pop_source() 538 */ 539 protected void push_source(@Nonnull Source source, boolean autopop) { 540 source.init(this); 541 source.setParent(this.source, autopop); 542 // source.setListener(listener); 543 if (listener != null) 544 listener.handleSourceChange(this.source, SourceChangeEvent.SUSPEND); 545 this.source = source; 546 if (listener != null) 547 listener.handleSourceChange(this.source, SourceChangeEvent.PUSH); 548 } 549 550 /** 551 * Pops a Source from the input stack. 552 * 553 * @see #getSource() 554 * @see #push_source(Source,boolean) 555 * 556 * @param linemarker TODO: currently ignored, might be a bug? 557 * @throws IOException if an I/O error occurs. 558 */ 559 @CheckForNull 560 protected Token pop_source(boolean linemarker) 561 throws IOException { 562 if (listener != null) 563 listener.handleSourceChange(this.source, SourceChangeEvent.POP); 564 Source s = this.source; 565 this.source = s.getParent(); 566 /* Always a noop unless called externally. */ 567 s.close(); 568 if (listener != null && this.source != null) 569 listener.handleSourceChange(this.source, SourceChangeEvent.RESUME); 570 571 Source t = getSource(); 572 if (getFeature(Feature.LINEMARKERS) 573 && s.isNumbered() 574 && t != null) { 575 /* We actually want 'did the nested source 576 * contain a newline token', which isNumbered() 577 * approximates. This is not perfect, but works. */ 578 return line_token(t.getLine(), t.getName(), " 2"); 579 } 580 581 return null; 582 } 583 584 protected void pop_source() 585 throws IOException { 586 pop_source(false); 587 } 588 589 @Nonnull 590 private Token next_source() { 591 if (inputs.isEmpty()) 592 return new Token(EOF); 593 Source s = inputs.remove(0); 594 push_source(s, true); 595 return line_token(s.getLine(), s.getName(), " 1"); 596 } 597 598 /* Source tokens */ 599 private Token source_token; 600 601 /* XXX Make this include the NL, and make all cpp directives eat 602 * their own NL. */ 603 @Nonnull 604 private Token line_token(int line, @CheckForNull String name, @Nonnull String extra) { 605 StringBuilder buf = new StringBuilder(); 606 buf.append("#line ").append(line) 607 .append(" \""); 608 /* XXX This call to escape(name) is correct but ugly. */ 609 if (name == null) 610 buf.append("<no file>"); 611 else 612 MacroTokenSource.escape(buf, name); 613 buf.append("\"").append(extra).append("\n"); 614 return new Token(P_LINE, line, 0, buf.toString(), null); 615 } 616 617 @Nonnull 618 private Token source_token() 619 throws IOException, 620 LexerException { 621 if (source_token != null) { 622 Token tok = source_token; 623 source_token = null; 624 if (getFeature(Feature.DEBUG)) 625 LOG.debug("Returning unget token " + tok); 626 return tok; 627 } 628 629 for (;;) { 630 Source s = getSource(); 631 if (s == null) { 632 Token t = next_source(); 633 if (t.getType() == P_LINE && !getFeature(Feature.LINEMARKERS)) 634 continue; 635 return t; 636 } 637 Token tok = s.token(); 638 /* XXX Refactor with skipline() */ 639 if (tok.getType() == EOF && s.isAutopop()) { 640 // System.out.println("Autopop " + s); 641 Token mark = pop_source(true); 642 if (mark != null) 643 return mark; 644 continue; 645 } 646 if (getFeature(Feature.DEBUG)) 647 LOG.debug("Returning fresh token " + tok); 648 return tok; 649 } 650 } 651 652 private void source_untoken(Token tok) { 653 if (this.source_token != null) 654 throw new IllegalStateException("Cannot return two tokens"); 655 this.source_token = tok; 656 } 657 658 private boolean isWhite(Token tok) { 659 int type = tok.getType(); 660 return (type == WHITESPACE) 661 || (type == CCOMMENT) 662 || (type == CPPCOMMENT); 663 } 664 665 private Token source_token_nonwhite() 666 throws IOException, 667 LexerException { 668 Token tok; 669 do { 670 tok = source_token(); 671 } while (isWhite(tok)); 672 return tok; 673 } 674 675 /** 676 * Returns an NL or an EOF token. 677 * 678 * The metadata on the token will be correct, which is better 679 * than generating a new one. 680 * 681 * This method can, as of recent patches, return a P_LINE token. 682 */ 683 private Token source_skipline(boolean white) 684 throws IOException, 685 LexerException { 686 // (new Exception("skipping line")).printStackTrace(System.out); 687 Source s = getSource(); 688 Token tok = s.skipline(white); 689 /* XXX Refactor with source_token() */ 690 if (tok.getType() == EOF && s.isAutopop()) { 691 // System.out.println("Autopop " + s); 692 Token mark = pop_source(true); 693 if (mark != null) 694 return mark; 695 } 696 return tok; 697 } 698 699 /* processes and expands a macro. */ 700 private boolean macro(Macro m, Token orig) 701 throws IOException, 702 LexerException { 703 Token tok; 704 List<Argument> args; 705 706 // System.out.println("pp: expanding " + m); 707 if (m.isFunctionLike()) { 708 OPEN: 709 for (;;) { 710 tok = source_token(); 711 // System.out.println("pp: open: token is " + tok); 712 switch (tok.getType()) { 713 case WHITESPACE: /* XXX Really? */ 714 715 case CCOMMENT: 716 case CPPCOMMENT: 717 case NL: 718 break; /* continue */ 719 720 case '(': 721 break OPEN; 722 default: 723 source_untoken(tok); 724 return false; 725 } 726 } 727 728 // tok = expanded_token_nonwhite(); 729 tok = source_token_nonwhite(); 730 731 /* We either have, or we should have args. 732 * This deals elegantly with the case that we have 733 * one empty arg. */ 734 if (tok.getType() != ')' || m.getArgs() > 0) { 735 args = new ArrayList<Argument>(); 736 737 Argument arg = new Argument(); 738 int depth = 0; 739 boolean space = false; 740 741 ARGS: 742 for (;;) { 743 // System.out.println("pp: arg: token is " + tok); 744 switch (tok.getType()) { 745 case EOF: 746 error(tok, "EOF in macro args"); 747 return false; 748 749 case ',': 750 if (depth == 0) { 751 if (m.isVariadic() 752 && /* We are building the last arg. */ args.size() == m.getArgs() - 1) { 753 /* Just add the comma. */ 754 arg.addToken(tok); 755 } else { 756 args.add(arg); 757 arg = new Argument(); 758 } 759 } else { 760 arg.addToken(tok); 761 } 762 space = false; 763 break; 764 case ')': 765 if (depth == 0) { 766 args.add(arg); 767 break ARGS; 768 } else { 769 depth--; 770 arg.addToken(tok); 771 } 772 space = false; 773 break; 774 case '(': 775 depth++; 776 arg.addToken(tok); 777 space = false; 778 break; 779 780 case WHITESPACE: 781 case CCOMMENT: 782 case CPPCOMMENT: 783 case NL: 784 /* Avoid duplicating spaces. */ 785 space = true; 786 break; 787 788 default: 789 /* Do not put space on the beginning of 790 * an argument token. */ 791 if (space && !arg.isEmpty()) 792 arg.addToken(Token.space); 793 arg.addToken(tok); 794 space = false; 795 break; 796 797 } 798 // tok = expanded_token(); 799 tok = source_token(); 800 } 801 /* space may still be true here, thus trailing space 802 * is stripped from arguments. */ 803 804 if (args.size() != m.getArgs()) { 805 if (m.isVariadic()) { 806 if (args.size() == m.getArgs() - 1) { 807 args.add(new Argument()); 808 } else { 809 error(tok, 810 "variadic macro " + m.getName() 811 + " has at least " + (m.getArgs() - 1) + " parameters " 812 + "but given " + args.size() + " args"); 813 return false; 814 } 815 } else { 816 error(tok, 817 "macro " + m.getName() 818 + " has " + m.getArgs() + " parameters " 819 + "but given " + args.size() + " args"); 820 /* We could replay the arg tokens, but I 821 * note that GNU cpp does exactly what we do, 822 * i.e. output the macro name and chew the args. 823 */ 824 return false; 825 } 826 } 827 828 for (Argument a : args) { 829 a.expand(this); 830 } 831 832 // System.out.println("Macro " + m + " args " + args); 833 } else { 834 /* nargs == 0 and we (correctly) got () */ 835 args = null; 836 } 837 838 } else { 839 /* Macro without args. */ 840 args = null; 841 } 842 843 if (m == __LINE__) { 844 push_source(new FixedTokenSource( 845 new Token[]{new Token(NUMBER, 846 orig.getLine(), orig.getColumn(), 847 Integer.toString(orig.getLine()), 848 new NumericValue(10, Integer.toString(orig.getLine())))} 849 ), true); 850 } else if (m == __FILE__) { 851 StringBuilder buf = new StringBuilder("\""); 852 String name = getSource().getName(); 853 if (name == null) 854 name = "<no file>"; 855 for (int i = 0; i < name.length(); i++) { 856 char c = name.charAt(i); 857 switch (c) { 858 case '\\': 859 buf.append("\\\\"); 860 break; 861 case '"': 862 buf.append("\\\""); 863 break; 864 default: 865 buf.append(c); 866 break; 867 } 868 } 869 buf.append("\""); 870 String text = buf.toString(); 871 push_source(new FixedTokenSource( 872 new Token[]{new Token(STRING, 873 orig.getLine(), orig.getColumn(), 874 text, text)} 875 ), true); 876 } else if (m == __COUNTER__) { 877 /* This could equivalently have been done by adding 878 * a special Macro subclass which overrides getTokens(). */ 879 int value = this.counter++; 880 push_source(new FixedTokenSource( 881 new Token[]{new Token(NUMBER, 882 orig.getLine(), orig.getColumn(), 883 Integer.toString(value), 884 new NumericValue(10, Integer.toString(value)))} 885 ), true); 886 } else { 887 push_source(new MacroTokenSource(m, args), true); 888 } 889 890 return true; 891 } 892 893 /** 894 * Expands an argument. 895 */ 896 /* I'd rather this were done lazily, but doing so breaks spec. */ 897 @Nonnull 898 /* pp */ List<Token> expand(@Nonnull List<Token> arg) 899 throws IOException, 900 LexerException { 901 List<Token> expansion = new ArrayList<Token>(); 902 boolean space = false; 903 904 push_source(new FixedTokenSource(arg), false); 905 906 EXPANSION: 907 for (;;) { 908 Token tok = expanded_token(); 909 switch (tok.getType()) { 910 case EOF: 911 break EXPANSION; 912 913 case WHITESPACE: 914 case CCOMMENT: 915 case CPPCOMMENT: 916 space = true; 917 break; 918 919 default: 920 if (space && !expansion.isEmpty()) 921 expansion.add(Token.space); 922 expansion.add(tok); 923 space = false; 924 break; 925 } 926 } 927 928 // Always returns null. 929 pop_source(false); 930 931 return expansion; 932 } 933 934 /* processes a #define directive */ 935 private Token define() 936 throws IOException, 937 LexerException { 938 Token tok = source_token_nonwhite(); 939 if (tok.getType() != IDENTIFIER) { 940 error(tok, "Expected identifier"); 941 return source_skipline(false); 942 } 943 /* if predefined */ 944 945 String name = tok.getText(); 946 if ("defined".equals(name)) { 947 error(tok, "Cannot redefine name 'defined'"); 948 return source_skipline(false); 949 } 950 951 Macro m = new Macro(getSource(), name); 952 List<String> args; 953 954 tok = source_token(); 955 if (tok.getType() == '(') { 956 tok = source_token_nonwhite(); 957 if (tok.getType() != ')') { 958 args = new ArrayList<String>(); 959 ARGS: 960 for (;;) { 961 switch (tok.getType()) { 962 case IDENTIFIER: 963 args.add(tok.getText()); 964 break; 965 case ELLIPSIS: 966 // Unnamed Variadic macro 967 args.add("__VA_ARGS__"); 968 // We just named the ellipsis, but we unget the token 969 // to allow the ELLIPSIS handling below to process it. 970 source_untoken(tok); 971 break; 972 case NL: 973 case EOF: 974 error(tok, 975 "Unterminated macro parameter list"); 976 return tok; 977 default: 978 error(tok, 979 "error in macro parameters: " 980 + tok.getText()); 981 return source_skipline(false); 982 } 983 tok = source_token_nonwhite(); 984 switch (tok.getType()) { 985 case ',': 986 break; 987 case ELLIPSIS: 988 tok = source_token_nonwhite(); 989 if (tok.getType() != ')') 990 error(tok, 991 "ellipsis must be on last argument"); 992 m.setVariadic(true); 993 break ARGS; 994 case ')': 995 break ARGS; 996 997 case NL: 998 case EOF: 999 /* Do not skip line. */ 1000 error(tok, 1001 "Unterminated macro parameters"); 1002 return tok; 1003 default: 1004 error(tok, 1005 "Bad token in macro parameters: " 1006 + tok.getText()); 1007 return source_skipline(false); 1008 } 1009 tok = source_token_nonwhite(); 1010 } 1011 } else { 1012 assert tok.getType() == ')' : "Expected ')'"; 1013 args = Collections.emptyList(); 1014 } 1015 1016 m.setArgs(args); 1017 } else { 1018 /* For searching. */ 1019 args = Collections.emptyList(); 1020 source_untoken(tok); 1021 } 1022 1023 /* Get an expansion for the macro, using indexOf. */ 1024 boolean space = false; 1025 boolean paste = false; 1026 int idx; 1027 1028 /* Ensure no space at start. */ 1029 tok = source_token_nonwhite(); 1030 EXPANSION: 1031 for (;;) { 1032 switch (tok.getType()) { 1033 case EOF: 1034 break EXPANSION; 1035 case NL: 1036 break EXPANSION; 1037 1038 case CCOMMENT: 1039 case CPPCOMMENT: 1040 /* XXX This is where we implement GNU's cpp -CC. */ 1041 // break; 1042 case WHITESPACE: 1043 if (!paste) 1044 space = true; 1045 break; 1046 1047 /* Paste. */ 1048 case PASTE: 1049 space = false; 1050 paste = true; 1051 m.addPaste(new Token(M_PASTE, 1052 tok.getLine(), tok.getColumn(), 1053 "#" + "#", null)); 1054 break; 1055 1056 /* Stringify. */ 1057 case '#': 1058 if (space) 1059 m.addToken(Token.space); 1060 space = false; 1061 Token la = source_token_nonwhite(); 1062 if (la.getType() == IDENTIFIER 1063 && ((idx = args.indexOf(la.getText())) != -1)) { 1064 m.addToken(new Token(M_STRING, 1065 la.getLine(), la.getColumn(), 1066 "#" + la.getText(), 1067 Integer.valueOf(idx))); 1068 } else { 1069 m.addToken(tok); 1070 /* Allow for special processing. */ 1071 source_untoken(la); 1072 } 1073 break; 1074 1075 case IDENTIFIER: 1076 if (space) 1077 m.addToken(Token.space); 1078 space = false; 1079 paste = false; 1080 idx = args.indexOf(tok.getText()); 1081 if (idx == -1) 1082 m.addToken(tok); 1083 else 1084 m.addToken(new Token(M_ARG, 1085 tok.getLine(), tok.getColumn(), 1086 tok.getText(), 1087 Integer.valueOf(idx))); 1088 break; 1089 1090 default: 1091 if (space) 1092 m.addToken(Token.space); 1093 space = false; 1094 paste = false; 1095 m.addToken(tok); 1096 break; 1097 } 1098 tok = source_token(); 1099 } 1100 1101 if (getFeature(Feature.DEBUG)) 1102 LOG.debug("Defined macro " + m); 1103 addMacro(m); 1104 1105 return tok; /* NL or EOF. */ 1106 1107 } 1108 1109 @Nonnull 1110 private Token undef() 1111 throws IOException, 1112 LexerException { 1113 Token tok = source_token_nonwhite(); 1114 if (tok.getType() != IDENTIFIER) { 1115 error(tok, 1116 "Expected identifier, not " + tok.getText()); 1117 if (tok.getType() == NL || tok.getType() == EOF) 1118 return tok; 1119 } else { 1120 Macro m = getMacro(tok.getText()); 1121 if (m != null) { 1122 /* XXX error if predefined */ 1123 macros.remove(m.getName()); 1124 } 1125 } 1126 return source_skipline(true); 1127 } 1128 1129 /** 1130 * Attempts to include the given file. 1131 * 1132 * User code may override this method to implement a virtual 1133 * file system. 1134 * 1135 * @param file The VirtualFile to attempt to include. 1136 * @return true if the file was successfully included, false otherwise. 1137 * @throws IOException if an I/O error occurs. 1138 */ 1139 protected boolean include(@Nonnull VirtualFile file) 1140 throws IOException { 1141 // System.out.println("Try to include " + ((File)file).getAbsolutePath()); 1142 if (!file.isFile()) 1143 return false; 1144 if (getFeature(Feature.DEBUG)) 1145 LOG.debug("pp: including " + file); 1146 includes.add(file); 1147 push_source(file.getSource(), true); 1148 return true; 1149 } 1150 1151 /** 1152 * Attempts to include a file from an include path, by name. 1153 * 1154 * @param path The list of virtual directories to search for the given name. 1155 * @param name The name of the file to attempt to include. 1156 * @return true if the file was successfully included, false otherwise. 1157 * @throws IOException if an I/O error occurs. 1158 */ 1159 protected boolean include(@Nonnull Iterable<String> path, @Nonnull String name) 1160 throws IOException { 1161 for (String dir : path) { 1162 VirtualFile file = getFileSystem().getFile(dir, name); 1163 if (include(file)) 1164 return true; 1165 } 1166 return false; 1167 } 1168 1169 /** 1170 * Handles an include directive. 1171 * 1172 * @throws IOException if an I/O error occurs. 1173 * @throws LexerException if the include fails, and the error handler is fatal. 1174 */ 1175 private void include( 1176 @CheckForNull String parent, int line, 1177 @Nonnull String name, boolean quoted, boolean next) 1178 throws IOException, 1179 LexerException { 1180 if (name.startsWith("/")) { 1181 VirtualFile file = filesystem.getFile(name); 1182 if (include(file)) 1183 return; 1184 StringBuilder buf = new StringBuilder(); 1185 buf.append("File not found: ").append(name); 1186 error(line, 0, buf.toString()); 1187 return; 1188 } 1189 1190 VirtualFile pdir = null; 1191 if (quoted) { 1192 if (parent != null) { 1193 VirtualFile pfile = filesystem.getFile(parent); 1194 pdir = pfile.getParentFile(); 1195 } 1196 if (pdir != null) { 1197 VirtualFile ifile = pdir.getChildFile(name); 1198 if (include(ifile)) 1199 return; 1200 } 1201 if (include(quoteincludepath, name)) 1202 return; 1203 } else { 1204 int idx = name.indexOf('/'); 1205 if (idx != -1) { 1206 String frameworkName = name.substring(0, idx); 1207 String headerName = name.substring(idx + 1); 1208 String headerPath = frameworkName + ".framework/Headers/" + headerName; 1209 if (include(frameworkspath, headerPath)) 1210 return; 1211 } 1212 } 1213 1214 if (include(sysincludepath, name)) 1215 return; 1216 1217 StringBuilder buf = new StringBuilder(); 1218 buf.append("File not found: ").append(name); 1219 buf.append(" in"); 1220 if (quoted) { 1221 buf.append(" .").append('(').append(pdir).append(')'); 1222 for (String dir : quoteincludepath) 1223 buf.append(" ").append(dir); 1224 } 1225 for (String dir : sysincludepath) 1226 buf.append(" ").append(dir); 1227 error(line, 0, buf.toString()); 1228 } 1229 1230 @Nonnull 1231 private Token include(boolean next) 1232 throws IOException, 1233 LexerException { 1234 LexerSource lexer = (LexerSource) source; 1235 try { 1236 lexer.setInclude(true); 1237 Token tok = token_nonwhite(); 1238 1239 String name; 1240 boolean quoted; 1241 1242 if (tok.getType() == STRING) { 1243 /* XXX Use the original text, not the value. 1244 * Backslashes must not be treated as escapes here. */ 1245 StringBuilder buf = new StringBuilder((String) tok.getValue()); 1246 HEADER: 1247 for (;;) { 1248 tok = token_nonwhite(); 1249 switch (tok.getType()) { 1250 case STRING: 1251 buf.append((String) tok.getValue()); 1252 break; 1253 case NL: 1254 case EOF: 1255 break HEADER; 1256 default: 1257 warning(tok, 1258 "Unexpected token on #" + "include line"); 1259 return source_skipline(false); 1260 } 1261 } 1262 name = buf.toString(); 1263 quoted = true; 1264 } else if (tok.getType() == HEADER) { 1265 name = (String) tok.getValue(); 1266 quoted = false; 1267 tok = source_skipline(true); 1268 } else { 1269 error(tok, 1270 "Expected string or header, not " + tok.getText()); 1271 switch (tok.getType()) { 1272 case NL: 1273 case EOF: 1274 return tok; 1275 default: 1276 /* Only if not a NL or EOF already. */ 1277 return source_skipline(false); 1278 } 1279 } 1280 1281 /* Do the inclusion. */ 1282 include(source.getPath(), tok.getLine(), name, quoted, next); 1283 1284 /* 'tok' is the 'nl' after the include. We use it after the 1285 * #line directive. */ 1286 if (getFeature(Feature.LINEMARKERS)) 1287 return line_token(1, source.getName(), " 1"); 1288 return tok; 1289 } finally { 1290 lexer.setInclude(false); 1291 } 1292 } 1293 1294 protected void pragma_once(@Nonnull Token name) 1295 throws IOException, LexerException { 1296 Source s = this.source; 1297 if (!onceseenpaths.add(s.getPath())) { 1298 Token mark = pop_source(true); 1299 // FixedTokenSource should never generate a linemarker on exit. 1300 if (mark != null) 1301 push_source(new FixedTokenSource(Arrays.asList(mark)), true); 1302 } 1303 } 1304 1305 protected void pragma(@Nonnull Token name, @Nonnull List<Token> value) 1306 throws IOException, 1307 LexerException { 1308 if (getFeature(Feature.PRAGMA_ONCE)) { 1309 if ("once".equals(name.getText())) { 1310 pragma_once(name); 1311 return; 1312 } 1313 } 1314 warning(name, "Unknown #" + "pragma: " + name.getText()); 1315 } 1316 1317 @Nonnull 1318 private Token pragma() 1319 throws IOException, 1320 LexerException { 1321 Token name; 1322 1323 NAME: 1324 for (;;) { 1325 Token tok = source_token(); 1326 switch (tok.getType()) { 1327 case EOF: 1328 /* There ought to be a newline before EOF. 1329 * At least, in any skipline context. */ 1330 /* XXX Are we sure about this? */ 1331 warning(tok, 1332 "End of file in #" + "pragma"); 1333 return tok; 1334 case NL: 1335 /* This may contain one or more newlines. */ 1336 warning(tok, 1337 "Empty #" + "pragma"); 1338 return tok; 1339 case CCOMMENT: 1340 case CPPCOMMENT: 1341 case WHITESPACE: 1342 continue NAME; 1343 case IDENTIFIER: 1344 name = tok; 1345 break NAME; 1346 default: 1347 warning(tok, 1348 "Illegal #" + "pragma " + tok.getText()); 1349 return source_skipline(false); 1350 } 1351 } 1352 1353 Token tok; 1354 List<Token> value = new ArrayList<Token>(); 1355 VALUE: 1356 for (;;) { 1357 tok = source_token(); 1358 switch (tok.getType()) { 1359 case EOF: 1360 /* There ought to be a newline before EOF. 1361 * At least, in any skipline context. */ 1362 /* XXX Are we sure about this? */ 1363 warning(tok, 1364 "End of file in #" + "pragma"); 1365 break VALUE; 1366 case NL: 1367 /* This may contain one or more newlines. */ 1368 break VALUE; 1369 case CCOMMENT: 1370 case CPPCOMMENT: 1371 break; 1372 case WHITESPACE: 1373 value.add(tok); 1374 break; 1375 default: 1376 value.add(tok); 1377 break; 1378 } 1379 } 1380 1381 pragma(name, value); 1382 1383 return tok; /* The NL. */ 1384 1385 } 1386 1387 /* For #error and #warning. */ 1388 private void error(@Nonnull Token pptok, boolean is_error) 1389 throws IOException, 1390 LexerException { 1391 StringBuilder buf = new StringBuilder(); 1392 buf.append('#').append(pptok.getText()).append(' '); 1393 /* Peculiar construction to ditch first whitespace. */ 1394 Token tok = source_token_nonwhite(); 1395 ERROR: 1396 for (;;) { 1397 switch (tok.getType()) { 1398 case NL: 1399 case EOF: 1400 break ERROR; 1401 default: 1402 buf.append(tok.getText()); 1403 break; 1404 } 1405 tok = source_token(); 1406 } 1407 if (is_error) 1408 error(pptok, buf.toString()); 1409 else 1410 warning(pptok, buf.toString()); 1411 } 1412 1413 /* This bypasses token() for #elif expressions. 1414 * If we don't do this, then isActive() == false 1415 * causes token() to simply chew the entire input line. */ 1416 @Nonnull 1417 private Token expanded_token() 1418 throws IOException, 1419 LexerException { 1420 for (;;) { 1421 Token tok = source_token(); 1422 // System.out.println("Source token is " + tok); 1423 if (tok.getType() == IDENTIFIER) { 1424 Macro m = getMacro(tok.getText()); 1425 if (m == null) 1426 return tok; 1427 if (source.isExpanding(m)) 1428 return tok; 1429 if (macro(m, tok)) 1430 continue; 1431 } 1432 return tok; 1433 } 1434 } 1435 1436 @Nonnull 1437 private Token expanded_token_nonwhite() 1438 throws IOException, 1439 LexerException { 1440 Token tok; 1441 do { 1442 tok = expanded_token(); 1443 // System.out.println("expanded token is " + tok); 1444 } while (isWhite(tok)); 1445 return tok; 1446 } 1447 1448 @CheckForNull 1449 private Token expr_token = null; 1450 1451 @Nonnull 1452 private Token expr_token() 1453 throws IOException, 1454 LexerException { 1455 Token tok = expr_token; 1456 1457 if (tok != null) { 1458 // System.out.println("ungetting"); 1459 expr_token = null; 1460 } else { 1461 tok = expanded_token_nonwhite(); 1462 // System.out.println("expt is " + tok); 1463 1464 if (tok.getType() == IDENTIFIER 1465 && tok.getText().equals("defined")) { 1466 Token la = source_token_nonwhite(); 1467 boolean paren = false; 1468 if (la.getType() == '(') { 1469 paren = true; 1470 la = source_token_nonwhite(); 1471 } 1472 1473 // System.out.println("Core token is " + la); 1474 if (la.getType() != IDENTIFIER) { 1475 error(la, 1476 "defined() needs identifier, not " 1477 + la.getText()); 1478 tok = new Token(NUMBER, 1479 la.getLine(), la.getColumn(), 1480 "0", new NumericValue(10, "0")); 1481 } else if (macros.containsKey(la.getText())) { 1482 // System.out.println("Found macro"); 1483 tok = new Token(NUMBER, 1484 la.getLine(), la.getColumn(), 1485 "1", new NumericValue(10, "1")); 1486 } else { 1487 // System.out.println("Not found macro"); 1488 tok = new Token(NUMBER, 1489 la.getLine(), la.getColumn(), 1490 "0", new NumericValue(10, "0")); 1491 } 1492 1493 if (paren) { 1494 la = source_token_nonwhite(); 1495 if (la.getType() != ')') { 1496 expr_untoken(la); 1497 error(la, "Missing ) in defined(). Got " + la.getText()); 1498 } 1499 } 1500 } 1501 } 1502 1503 // System.out.println("expr_token returns " + tok); 1504 return tok; 1505 } 1506 1507 private void expr_untoken(@Nonnull Token tok) 1508 throws LexerException { 1509 if (expr_token != null) 1510 throw new InternalException( 1511 "Cannot unget two expression tokens." 1512 ); 1513 expr_token = tok; 1514 } 1515 1516 private int expr_priority(@Nonnull Token op) { 1517 switch (op.getType()) { 1518 case '/': 1519 return 11; 1520 case '%': 1521 return 11; 1522 case '*': 1523 return 11; 1524 case '+': 1525 return 10; 1526 case '-': 1527 return 10; 1528 case LSH: 1529 return 9; 1530 case RSH: 1531 return 9; 1532 case '<': 1533 return 8; 1534 case '>': 1535 return 8; 1536 case LE: 1537 return 8; 1538 case GE: 1539 return 8; 1540 case EQ: 1541 return 7; 1542 case NE: 1543 return 7; 1544 case '&': 1545 return 6; 1546 case '^': 1547 return 5; 1548 case '|': 1549 return 4; 1550 case LAND: 1551 return 3; 1552 case LOR: 1553 return 2; 1554 case '?': 1555 return 1; 1556 default: 1557 // System.out.println("Unrecognised operator " + op); 1558 return 0; 1559 } 1560 } 1561 1562 private int expr_char(Token token) { 1563 Object value = token.getValue(); 1564 if (value instanceof Character) 1565 return ((Character) value).charValue(); 1566 String text = String.valueOf(value); 1567 if (text.length() == 0) 1568 return 0; 1569 return text.charAt(0); 1570 } 1571 1572 private long expr(int priority) 1573 throws IOException, 1574 LexerException { 1575 /* 1576 * (new Exception("expr(" + priority + ") called")).printStackTrace(); 1577 */ 1578 1579 Token tok = expr_token(); 1580 long lhs, rhs; 1581 1582 // System.out.println("Expr lhs token is " + tok); 1583 switch (tok.getType()) { 1584 case '(': 1585 lhs = expr(0); 1586 tok = expr_token(); 1587 if (tok.getType() != ')') { 1588 expr_untoken(tok); 1589 error(tok, "Missing ) in expression. Got " + tok.getText()); 1590 return 0; 1591 } 1592 break; 1593 1594 case '~': 1595 lhs = ~expr(11); 1596 break; 1597 case '!': 1598 lhs = expr(11) == 0 ? 1 : 0; 1599 break; 1600 case '-': 1601 lhs = -expr(11); 1602 break; 1603 case NUMBER: 1604 NumericValue value = (NumericValue) tok.getValue(); 1605 lhs = value.longValue(); 1606 break; 1607 case CHARACTER: 1608 lhs = expr_char(tok); 1609 break; 1610 case IDENTIFIER: 1611 if (warnings.contains(Warning.UNDEF)) 1612 warning(tok, "Undefined token '" + tok.getText() 1613 + "' encountered in conditional."); 1614 lhs = 0; 1615 break; 1616 1617 default: 1618 expr_untoken(tok); 1619 error(tok, 1620 "Bad token in expression: " + tok.getText()); 1621 return 0; 1622 } 1623 1624 EXPR: 1625 for (;;) { 1626 // System.out.println("expr: lhs is " + lhs + ", pri = " + priority); 1627 Token op = expr_token(); 1628 int pri = expr_priority(op); /* 0 if not a binop. */ 1629 1630 if (pri == 0 || priority >= pri) { 1631 expr_untoken(op); 1632 break EXPR; 1633 } 1634 rhs = expr(pri); 1635 // System.out.println("rhs token is " + rhs); 1636 switch (op.getType()) { 1637 case '/': 1638 if (rhs == 0) { 1639 error(op, "Division by zero"); 1640 lhs = 0; 1641 } else { 1642 lhs = lhs / rhs; 1643 } 1644 break; 1645 case '%': 1646 if (rhs == 0) { 1647 error(op, "Modulus by zero"); 1648 lhs = 0; 1649 } else { 1650 lhs = lhs % rhs; 1651 } 1652 break; 1653 case '*': 1654 lhs = lhs * rhs; 1655 break; 1656 case '+': 1657 lhs = lhs + rhs; 1658 break; 1659 case '-': 1660 lhs = lhs - rhs; 1661 break; 1662 case '<': 1663 lhs = lhs < rhs ? 1 : 0; 1664 break; 1665 case '>': 1666 lhs = lhs > rhs ? 1 : 0; 1667 break; 1668 case '&': 1669 lhs = lhs & rhs; 1670 break; 1671 case '^': 1672 lhs = lhs ^ rhs; 1673 break; 1674 case '|': 1675 lhs = lhs | rhs; 1676 break; 1677 1678 case LSH: 1679 lhs = lhs << rhs; 1680 break; 1681 case RSH: 1682 lhs = lhs >> rhs; 1683 break; 1684 case LE: 1685 lhs = lhs <= rhs ? 1 : 0; 1686 break; 1687 case GE: 1688 lhs = lhs >= rhs ? 1 : 0; 1689 break; 1690 case EQ: 1691 lhs = lhs == rhs ? 1 : 0; 1692 break; 1693 case NE: 1694 lhs = lhs != rhs ? 1 : 0; 1695 break; 1696 case LAND: 1697 lhs = (lhs != 0) && (rhs != 0) ? 1 : 0; 1698 break; 1699 case LOR: 1700 lhs = (lhs != 0) || (rhs != 0) ? 1 : 0; 1701 break; 1702 1703 case '?': { 1704 tok = expr_token(); 1705 if (tok.getType() != ':') { 1706 expr_untoken(tok); 1707 error(tok, "Missing : in conditional expression. Got " + tok.getText()); 1708 return 0; 1709 } 1710 long falseResult = expr(0); 1711 lhs = (lhs != 0) ? rhs : falseResult; 1712 } 1713 break; 1714 1715 default: 1716 error(op, 1717 "Unexpected operator " + op.getText()); 1718 return 0; 1719 1720 } 1721 } 1722 1723 /* 1724 * (new Exception("expr returning " + lhs)).printStackTrace(); 1725 */ 1726 // System.out.println("expr returning " + lhs); 1727 return lhs; 1728 } 1729 1730 @Nonnull 1731 private Token toWhitespace(@Nonnull Token tok) { 1732 String text = tok.getText(); 1733 int len = text.length(); 1734 boolean cr = false; 1735 int nls = 0; 1736 1737 for (int i = 0; i < len; i++) { 1738 char c = text.charAt(i); 1739 1740 switch (c) { 1741 case '\r': 1742 cr = true; 1743 nls++; 1744 break; 1745 case '\n': 1746 if (cr) { 1747 cr = false; 1748 break; 1749 } 1750 /* fallthrough */ 1751 case '\u2028': 1752 case '\u2029': 1753 case '\u000B': 1754 case '\u000C': 1755 case '\u0085': 1756 cr = false; 1757 nls++; 1758 break; 1759 } 1760 } 1761 1762 char[] cbuf = new char[nls]; 1763 Arrays.fill(cbuf, '\n'); 1764 return new Token(WHITESPACE, 1765 tok.getLine(), tok.getColumn(), 1766 new String(cbuf)); 1767 } 1768 1769 @Nonnull 1770 private Token _token() 1771 throws IOException, 1772 LexerException { 1773 1774 for (;;) { 1775 Token tok; 1776 if (!isActive()) { 1777 Source s = getSource(); 1778 if (s == null) { 1779 Token t = next_source(); 1780 if (t.getType() == P_LINE && !getFeature(Feature.LINEMARKERS)) 1781 continue; 1782 return t; 1783 } 1784 1785 try { 1786 /* XXX Tell lexer to ignore warnings. */ 1787 s.setActive(false); 1788 tok = source_token(); 1789 } finally { 1790 /* XXX Tell lexer to stop ignoring warnings. */ 1791 s.setActive(true); 1792 } 1793 switch (tok.getType()) { 1794 case HASH: 1795 case NL: 1796 case EOF: 1797 /* The preprocessor has to take action here. */ 1798 break; 1799 case WHITESPACE: 1800 return tok; 1801 case CCOMMENT: 1802 case CPPCOMMENT: 1803 // Patch up to preserve whitespace. 1804 if (getFeature(Feature.KEEPALLCOMMENTS)) 1805 return tok; 1806 if (!isActive()) 1807 return toWhitespace(tok); 1808 if (getFeature(Feature.KEEPCOMMENTS)) 1809 return tok; 1810 return toWhitespace(tok); 1811 default: 1812 // Return NL to preserve whitespace. 1813 /* XXX This might lose a comment. */ 1814 return source_skipline(false); 1815 } 1816 } else { 1817 tok = source_token(); 1818 } 1819 1820 LEX: 1821 switch (tok.getType()) { 1822 case EOF: 1823 /* Pop the stacks. */ 1824 return tok; 1825 1826 case WHITESPACE: 1827 case NL: 1828 return tok; 1829 1830 case CCOMMENT: 1831 case CPPCOMMENT: 1832 return tok; 1833 1834 case '!': 1835 case '%': 1836 case '&': 1837 case '(': 1838 case ')': 1839 case '*': 1840 case '+': 1841 case ',': 1842 case '-': 1843 case '/': 1844 case ':': 1845 case ';': 1846 case '<': 1847 case '=': 1848 case '>': 1849 case '?': 1850 case '[': 1851 case ']': 1852 case '^': 1853 case '{': 1854 case '|': 1855 case '}': 1856 case '~': 1857 case '.': 1858 1859 /* From Olivier Chafik for Objective C? */ 1860 case '@': 1861 /* The one remaining ASCII, might as well. */ 1862 case '`': 1863 1864 // case '#': 1865 case AND_EQ: 1866 case ARROW: 1867 case CHARACTER: 1868 case DEC: 1869 case DIV_EQ: 1870 case ELLIPSIS: 1871 case EQ: 1872 case GE: 1873 case HEADER: /* Should only arise from include() */ 1874 1875 case INC: 1876 case LAND: 1877 case LE: 1878 case LOR: 1879 case LSH: 1880 case LSH_EQ: 1881 case SUB_EQ: 1882 case MOD_EQ: 1883 case MULT_EQ: 1884 case NE: 1885 case OR_EQ: 1886 case PLUS_EQ: 1887 case RANGE: 1888 case RSH: 1889 case RSH_EQ: 1890 case STRING: 1891 case SQSTRING: 1892 case XOR_EQ: 1893 return tok; 1894 1895 case NUMBER: 1896 return tok; 1897 1898 case IDENTIFIER: 1899 Macro m = getMacro(tok.getText()); 1900 if (m == null) 1901 return tok; 1902 if (source.isExpanding(m)) 1903 return tok; 1904 if (macro(m, tok)) 1905 break; 1906 return tok; 1907 1908 case P_LINE: 1909 if (getFeature(Feature.LINEMARKERS)) 1910 return tok; 1911 break; 1912 1913 case INVALID: 1914 if (getFeature(Feature.CSYNTAX)) 1915 error(tok, String.valueOf(tok.getValue())); 1916 return tok; 1917 1918 default: 1919 throw new InternalException("Bad token " + tok); 1920 // break; 1921 1922 case HASH: 1923 tok = source_token_nonwhite(); 1924 // (new Exception("here")).printStackTrace(); 1925 switch (tok.getType()) { 1926 case NL: 1927 break LEX; /* Some code has #\n */ 1928 1929 case IDENTIFIER: 1930 break; 1931 default: 1932 error(tok, 1933 "Preprocessor directive not a word " 1934 + tok.getText()); 1935 return source_skipline(false); 1936 } 1937 PreprocessorCommand ppcmd = PreprocessorCommand.forText(tok.getText()); 1938 if (ppcmd == null) { 1939 error(tok, 1940 "Unknown preprocessor directive " 1941 + tok.getText()); 1942 return source_skipline(false); 1943 } 1944 1945 PP: 1946 switch (ppcmd) { 1947 1948 case PP_DEFINE: 1949 if (!isActive()) 1950 return source_skipline(false); 1951 else 1952 return define(); 1953 // break; 1954 1955 case PP_UNDEF: 1956 if (!isActive()) 1957 return source_skipline(false); 1958 else 1959 return undef(); 1960 // break; 1961 1962 case PP_INCLUDE: 1963 if (!isActive()) 1964 return source_skipline(false); 1965 else 1966 return include(false); 1967 // break; 1968 case PP_INCLUDE_NEXT: 1969 if (!isActive()) 1970 return source_skipline(false); 1971 if (!getFeature(Feature.INCLUDENEXT)) { 1972 error(tok, 1973 "Directive include_next not enabled" 1974 ); 1975 return source_skipline(false); 1976 } 1977 return include(true); 1978 // break; 1979 1980 case PP_WARNING: 1981 case PP_ERROR: 1982 if (!isActive()) 1983 return source_skipline(false); 1984 else 1985 error(tok, ppcmd == PP_ERROR); 1986 break; 1987 1988 case PP_IF: 1989 push_state(); 1990 if (!isActive()) { 1991 return source_skipline(false); 1992 } 1993 expr_token = null; 1994 states.peek().setActive(expr(0) != 0); 1995 tok = expr_token(); /* unget */ 1996 1997 if (tok.getType() == NL) 1998 return tok; 1999 return source_skipline(true); 2000 // break; 2001 2002 case PP_ELIF: 2003 State state = states.peek(); 2004 if (false) { 2005 /* Check for 'if' */; 2006 } else if (state.sawElse()) { 2007 error(tok, 2008 "#elif after #" + "else"); 2009 return source_skipline(false); 2010 } else if (!state.isParentActive()) { 2011 /* Nested in skipped 'if' */ 2012 return source_skipline(false); 2013 } else if (state.isActive()) { 2014 /* The 'if' part got executed. */ 2015 state.setParentActive(false); 2016 /* This is like # else # if but with 2017 * only one # end. */ 2018 state.setActive(false); 2019 return source_skipline(false); 2020 } else { 2021 expr_token = null; 2022 state.setActive(expr(0) != 0); 2023 tok = expr_token(); /* unget */ 2024 2025 if (tok.getType() == NL) 2026 return tok; 2027 return source_skipline(true); 2028 } 2029 // break; 2030 2031 case PP_ELSE: 2032 state = states.peek(); 2033 if (false) 2034 /* Check for 'if' */ ; else if (state.sawElse()) { 2035 error(tok, 2036 "#" + "else after #" + "else"); 2037 return source_skipline(false); 2038 } else { 2039 state.setSawElse(); 2040 state.setActive(!state.isActive()); 2041 return source_skipline(warnings.contains(Warning.ENDIF_LABELS)); 2042 } 2043 // break; 2044 2045 case PP_IFDEF: 2046 push_state(); 2047 if (!isActive()) { 2048 return source_skipline(false); 2049 } else { 2050 tok = source_token_nonwhite(); 2051 // System.out.println("ifdef " + tok); 2052 if (tok.getType() != IDENTIFIER) { 2053 error(tok, 2054 "Expected identifier, not " 2055 + tok.getText()); 2056 return source_skipline(false); 2057 } else { 2058 String text = tok.getText(); 2059 boolean exists 2060 = macros.containsKey(text); 2061 states.peek().setActive(exists); 2062 return source_skipline(true); 2063 } 2064 } 2065 // break; 2066 2067 case PP_IFNDEF: 2068 push_state(); 2069 if (!isActive()) { 2070 return source_skipline(false); 2071 } else { 2072 tok = source_token_nonwhite(); 2073 if (tok.getType() != IDENTIFIER) { 2074 error(tok, 2075 "Expected identifier, not " 2076 + tok.getText()); 2077 return source_skipline(false); 2078 } else { 2079 String text = tok.getText(); 2080 boolean exists 2081 = macros.containsKey(text); 2082 states.peek().setActive(!exists); 2083 return source_skipline(true); 2084 } 2085 } 2086 // break; 2087 2088 case PP_ENDIF: 2089 pop_state(); 2090 return source_skipline(warnings.contains(Warning.ENDIF_LABELS)); 2091 // break; 2092 2093 case PP_LINE: 2094 return source_skipline(false); 2095 // break; 2096 2097 case PP_PRAGMA: 2098 if (!isActive()) 2099 return source_skipline(false); 2100 return pragma(); 2101 // break; 2102 2103 default: 2104 /* Actual unknown directives are 2105 * processed above. If we get here, 2106 * we succeeded the map lookup but 2107 * failed to handle it. Therefore, 2108 * this is (unconditionally?) fatal. */ 2109 // if (isActive()) /* XXX Could be warning. */ 2110 throw new InternalException( 2111 "Internal error: Unknown directive " 2112 + tok); 2113 // return source_skipline(false); 2114 } 2115 2116 } 2117 } 2118 } 2119 2120 @Nonnull 2121 private Token token_nonwhite() 2122 throws IOException, 2123 LexerException { 2124 Token tok; 2125 do { 2126 tok = _token(); 2127 } while (isWhite(tok)); 2128 return tok; 2129 } 2130 2131 /** 2132 * Returns the next preprocessor token. 2133 * 2134 * @see Token 2135 * @return The next fully preprocessed token. 2136 * @throws IOException if an I/O error occurs. 2137 * @throws LexerException if a preprocessing error occurs. 2138 * @throws InternalException if an unexpected error condition arises. 2139 */ 2140 @Nonnull 2141 public Token token() 2142 throws IOException, 2143 LexerException { 2144 Token tok = _token(); 2145 if (getFeature(Feature.DEBUG)) 2146 LOG.debug("pp: Returning " + tok); 2147 return tok; 2148 } 2149 2150 @Override 2151 public String toString() { 2152 StringBuilder buf = new StringBuilder(); 2153 2154 Source s = getSource(); 2155 while (s != null) { 2156 buf.append(" -> ").append(String.valueOf(s)).append("\n"); 2157 s = s.getParent(); 2158 } 2159 2160 Map<String, Macro> macros = new TreeMap<String, Macro>(getMacros()); 2161 for (Macro macro : macros.values()) { 2162 buf.append("#").append("macro ").append(macro).append("\n"); 2163 } 2164 2165 return buf.toString(); 2166 } 2167 2168 @Override 2169 public void close() 2170 throws IOException { 2171 { 2172 Source s = source; 2173 while (s != null) { 2174 s.close(); 2175 s = s.getParent(); 2176 } 2177 } 2178 for (Source s : inputs) { 2179 s.close(); 2180 } 2181 } 2182 2183}