Qt
Internal/Contributor docs for the Qt SDK. Note: These are NOT official API docs; those are found at https://doc.qt.io/
Loading...
Searching...
No Matches
preprocessor.cpp
Go to the documentation of this file.
1// Copyright (C) 2016 The Qt Company Ltd.
2// Copyright (C) 2014 Olivier Goffart <ogoffart@woboq.org>
3// SPDX-License-Identifier: LicenseRef-Qt-Commercial OR GPL-3.0-only WITH Qt-GPL-exception-1.0
4
5#include "preprocessor.h"
6#include "utils.h"
7#include <qstringlist.h>
8#include <qfile.h>
9#include <qdir.h>
10#include <qfileinfo.h>
11#include <qvarlengtharray.h>
12
14
15using namespace QtMiscUtils;
16
17#include "ppkeywords.cpp"
18#include "keywords.cpp"
19
20// transform \r\n into \n
21// \r into \n (os9 style)
22// backslash-newlines into newlines
23static QByteArray cleaned(const QByteArray &input)
24{
25 QByteArray result;
26 result.resize(input.size());
27 const char *data = input.constData();
28 const char *end = input.constData() + input.size();
29 char *output = result.data();
30
31 int newlines = 0;
32 while (data != end) {
33 while (data != end && is_space(*data))
34 ++data;
35 bool takeLine = (*data == '#');
36 if (*data == '%' && *(data+1) == ':') {
37 takeLine = true;
38 ++data;
39 }
40 if (takeLine) {
41 *output = '#';
42 ++output;
43 do ++data; while (data != end && is_space(*data));
44 }
45 while (data != end) {
46 // handle \\\n, \\\r\n and \\\r
47 if (*data == '\\') {
48 if (*(data + 1) == '\r') {
49 ++data;
50 }
51 if (data != end && (*(data + 1) == '\n' || (*data) == '\r')) {
52 ++newlines;
53 data += 1;
54 if (data != end && *data != '\r')
55 data += 1;
56 continue;
57 }
58 } else if (*data == '\r' && *(data + 1) == '\n') { // reduce \r\n to \n
59 ++data;
60 }
61 if (data == end)
62 break;
63
64 char ch = *data;
65 if (ch == '\r') // os9: replace \r with \n
66 ch = '\n';
67 *output = ch;
68 ++output;
69
70 if (*data == '\n') {
71 // output additional newlines to keep the correct line-numbering
72 // for the lines following the backslash-newline sequence(s)
73 while (newlines) {
74 *output = '\n';
75 ++output;
76 --newlines;
77 }
78 ++data;
79 break;
80 }
81 ++data;
82 }
83 }
84 result.resize(output - result.constData());
85 return result;
86}
87
88bool Preprocessor::preprocessOnly = false;
90{
91 while(index < symbols.size() - 1 && symbols.at(index).token != PP_ENDIF){
92 switch (symbols.at(index).token) {
93 case PP_IF:
94 case PP_IFDEF:
95 case PP_IFNDEF:
96 ++index;
98 break;
99 default:
100 ;
101 }
102 ++index;
103 }
104}
105
107{
108 while (index < symbols.size() - 1
109 && (symbols.at(index).token != PP_ENDIF
110 && symbols.at(index).token != PP_ELIF
111 && symbols.at(index).token != PP_ELSE)
112 ){
113 switch (symbols.at(index).token) {
114 case PP_IF:
115 case PP_IFDEF:
116 case PP_IFNDEF:
117 ++index;
119 break;
120 default:
121 ;
122 }
123 ++index;
124 }
125 return (index < symbols.size() - 1);
126}
127
128
129Symbols Preprocessor::tokenize(const QByteArray& input, int lineNum, Preprocessor::TokenizeMode mode)
130{
131 Symbols symbols;
132 // Preallocate some space to speed up the code below.
133 // The magic divisor value was found by calculating the average ratio between
134 // input size and the final size of symbols.
135 // This yielded a value of 16.x when compiling Qt Base.
136 symbols.reserve(input.size() / 16);
137 const char *begin = input.constData();
138 const char *data = begin;
139 while (*data) {
140 if (mode == TokenizeCpp || mode == TokenizeDefine) {
141 int column = 0;
142
143 const char *lexem = data;
144 int state = 0;
145 Token token = NOTOKEN;
146 for (;;) {
147 if (static_cast<signed char>(*data) < 0) {
148 ++data;
149 continue;
150 }
151 int nextindex = keywords[state].next;
152 int next = 0;
153 if (*data == keywords[state].defchar)
154 next = keywords[state].defnext;
155 else if (!state || nextindex)
156 next = keyword_trans[nextindex][(int)*data];
157 if (!next)
158 break;
159 state = next;
160 token = keywords[state].token;
161 ++data;
162 }
163
164 // suboptimal, is_ident_char should use a table
165 if (keywords[state].ident && is_ident_char(*data))
166 token = keywords[state].ident;
167
168 if (token == NOTOKEN) {
169 if (*data)
170 ++data;
171 // an error really, but let's ignore this input
172 // to not confuse moc later. However in pre-processor
173 // only mode let's continue.
175 continue;
176 }
177
178 ++column;
179
180 if (token > SPECIAL_TREATMENT_MARK) {
181 switch (token) {
182 case QUOTE:
183 data = skipQuote(data);
184 token = STRING_LITERAL;
185 // concatenate multi-line strings for easier
186 // STRING_LITERAL handling in moc
188 && !symbols.isEmpty()
189 && symbols.constLast().token == STRING_LITERAL) {
190
191 const QByteArray newString
192 = '\"'
193 + symbols.constLast().unquotedLexemView()
194 + input.mid(lexem - begin + 1, data - lexem - 2)
195 + '\"';
196 symbols.last() = Symbol(symbols.constLast().lineNum,
197 STRING_LITERAL,
198 newString);
199 continue;
200 }
201 break;
202 case SINGLEQUOTE:
203 while (*data && (*data != '\''
204 || (*(data-1)=='\\'
205 && *(data-2)!='\\')))
206 ++data;
207 if (*data)
208 ++data;
209 token = CHARACTER_LITERAL;
210 break;
211 case LANGLE_SCOPE:
212 // split <:: into two tokens, < and ::
213 token = LANGLE;
214 data -= 2;
215 break;
216 case DIGIT:
217 {
218 bool hasSeenTokenSeparator = false;;
219 while (isAsciiDigit(*data) || (hasSeenTokenSeparator = *data == '\''))
220 ++data;
221 if (!*data || *data != '.') {
222 token = INTEGER_LITERAL;
223 if (data - lexem == 1 &&
224 (*data == 'x' || *data == 'X'
225 || *data == 'b' || *data == 'B')
226 && *lexem == '0') {
227 ++data;
228 while (isHexDigit(*data) || (hasSeenTokenSeparator = *data == '\''))
229 ++data;
230 } else if (*data == 'L') // TODO: handle other suffixes
231 ++data;
232 if (!hasSeenTokenSeparator) {
233 while (is_ident_char(*data)) {
234 ++data;
235 token = IDENTIFIER;
236 }
237 }
238 break;
239 }
240 token = FLOATING_LITERAL;
241 ++data;
242 Q_FALLTHROUGH();
243 }
244 case FLOATING_LITERAL:
245 while (isAsciiDigit(*data) || *data == '\'')
246 ++data;
247 if (*data == '+' || *data == '-')
248 ++data;
249 if (*data == 'e' || *data == 'E') {
250 ++data;
251 while (isAsciiDigit(*data) || *data == '\'')
252 ++data;
253 }
254 if (*data == 'f' || *data == 'F'
255 || *data == 'l' || *data == 'L')
256 ++data;
257 break;
258 case HASH:
259 if (column == 1 && mode == TokenizeCpp) {
261 while (*data && (*data == ' ' || *data == '\t'))
262 ++data;
263 if (is_ident_char(*data))
265 continue;
266 }
267 break;
268 case PP_HASHHASH:
269 if (mode == TokenizeCpp)
270 continue;
271 break;
272 case NEWLINE:
273 ++lineNum;
274 if (mode == TokenizeDefine) {
275 mode = TokenizeCpp;
276 // emit the newline token
277 break;
278 }
279 continue;
280 case BACKSLASH:
281 {
282 const char *rewind = data;
283 while (*data && (*data == ' ' || *data == '\t'))
284 ++data;
285 if (*data && *data == '\n') {
286 ++data;
287 continue;
288 }
289 data = rewind;
290 } break;
291 case CHARACTER:
292 while (is_ident_char(*data))
293 ++data;
294 token = IDENTIFIER;
295 break;
296 case C_COMMENT:
297 if (*data) {
298 if (*data == '\n')
299 ++lineNum;
300 ++data;
301 if (*data) {
302 if (*data == '\n')
303 ++lineNum;
304 ++data;
305 }
306 }
307 while (*data && (*(data-1) != '/' || *(data-2) != '*')) {
308 if (*data == '\n')
309 ++lineNum;
310 ++data;
311 }
312 token = WHITESPACE; // one comment, one whitespace
313 Q_FALLTHROUGH();
314 case WHITESPACE:
315 if (column == 1)
316 column = 0;
317 while (*data && (*data == ' ' || *data == '\t'))
318 ++data;
319 if (Preprocessor::preprocessOnly) // tokenize whitespace
320 break;
321 continue;
322 case CPP_COMMENT:
323 while (*data && *data != '\n')
324 ++data;
325 continue; // ignore safely, the newline is a separator
326 default:
327 continue; //ignore
328 }
329 }
330 symbols += Symbol(lineNum, token, input, lexem-begin, data-lexem);
331
332 } else { // Preprocessor
333
334 const char *lexem = data;
335 int state = 0;
336 Token token = NOTOKEN;
337 if (mode == TokenizePreprocessorStatement) {
338 state = pp_keyword_trans[0][(int)'#'];
340 }
341 for (;;) {
342 if (static_cast<signed char>(*data) < 0) {
343 ++data;
344 continue;
345 }
346 int nextindex = pp_keywords[state].next;
347 int next = 0;
348 if (*data == pp_keywords[state].defchar)
349 next = pp_keywords[state].defnext;
350 else if (!state || nextindex)
351 next = pp_keyword_trans[nextindex][(int)*data];
352 if (!next)
353 break;
354 state = next;
355 token = pp_keywords[state].token;
356 ++data;
357 }
358 // suboptimal, is_ident_char should use a table
359 if (pp_keywords[state].ident && is_ident_char(*data))
360 token = pp_keywords[state].ident;
361
362 switch (token) {
363 case NOTOKEN:
364 if (*data)
365 ++data;
366 break;
367 case PP_DEFINE:
368 mode = PrepareDefine;
369 break;
370 case PP_IFDEF:
371 symbols += Symbol(lineNum, PP_IF);
372 symbols += Symbol(lineNum, PP_DEFINED);
373 continue;
374 case PP_IFNDEF:
375 symbols += Symbol(lineNum, PP_IF);
376 symbols += Symbol(lineNum, PP_NOT);
377 symbols += Symbol(lineNum, PP_DEFINED);
378 continue;
379 case PP_INCLUDE:
380 case PP_INCLUDE_NEXT:
381 mode = TokenizeInclude;
382 break;
383 case PP_QUOTE:
384 data = skipQuote(data);
385 token = PP_STRING_LITERAL;
386 break;
387 case PP_SINGLEQUOTE:
388 while (*data && (*data != '\''
389 || (*(data-1)=='\\'
390 && *(data-2)!='\\')))
391 ++data;
392 if (*data)
393 ++data;
394 token = PP_CHARACTER_LITERAL;
395 break;
396 case PP_DIGIT:
397 while (isAsciiDigit(*data) || *data == '\'')
398 ++data;
399 if (!*data || *data != '.') {
400 token = PP_INTEGER_LITERAL;
401 if (data - lexem == 1 &&
402 (*data == 'x' || *data == 'X')
403 && *lexem == '0') {
404 ++data;
405 while (isHexDigit(*data) || *data == '\'')
406 ++data;
407 } else if (*data == 'L') // TODO: handle other suffixes
408 ++data;
409 break;
410 }
411 token = PP_FLOATING_LITERAL;
412 ++data;
413 Q_FALLTHROUGH();
414 case PP_FLOATING_LITERAL:
415 while (isAsciiDigit(*data) || *data == '\'')
416 ++data;
417 if (*data == '+' || *data == '-')
418 ++data;
419 if (*data == 'e' || *data == 'E') {
420 ++data;
421 while (isAsciiDigit(*data) || *data == '\'')
422 ++data;
423 }
424 if (*data == 'f' || *data == 'F'
425 || *data == 'l' || *data == 'L')
426 ++data;
427 break;
428 case PP_CHARACTER:
429 if (mode == PreparePreprocessorStatement) {
430 // rewind entire token to begin
431 data = lexem;
433 continue;
434 }
435 while (is_ident_char(*data))
436 ++data;
437 token = PP_IDENTIFIER;
438
439 if (mode == PrepareDefine) {
440 symbols += Symbol(lineNum, token, input, lexem-begin, data-lexem);
441 // make sure we explicitly add the whitespace here if the next char
442 // is not an opening brace, so we can distinguish correctly between
443 // regular and function macros
444 if (*data != '(')
445 symbols += Symbol(lineNum, WHITESPACE);
446 mode = TokenizeDefine;
447 continue;
448 }
449 break;
450 case PP_C_COMMENT:
451 if (*data) {
452 if (*data == '\n')
453 ++lineNum;
454 ++data;
455 if (*data) {
456 if (*data == '\n')
457 ++lineNum;
458 ++data;
459 }
460 }
461 while (*data && (*(data-1) != '/' || *(data-2) != '*')) {
462 if (*data == '\n')
463 ++lineNum;
464 ++data;
465 }
466 token = PP_WHITESPACE; // one comment, one whitespace
467 Q_FALLTHROUGH();
468 case PP_WHITESPACE:
469 while (*data && (*data == ' ' || *data == '\t'))
470 ++data;
471 continue; // the preprocessor needs no whitespace
472 case PP_CPP_COMMENT:
473 while (*data && *data != '\n')
474 ++data;
475 continue; // ignore safely, the newline is a separator
476 case PP_NEWLINE:
477 ++lineNum;
478 mode = TokenizeCpp;
479 break;
480 case PP_BACKSLASH:
481 {
482 const char *rewind = data;
483 while (*data && (*data == ' ' || *data == '\t'))
484 ++data;
485 if (*data && *data == '\n') {
486 ++data;
487 continue;
488 }
489 data = rewind;
490 } break;
491 case PP_LANGLE:
492 if (mode != TokenizeInclude)
493 break;
494 token = PP_STRING_LITERAL;
495 while (*data && *data != '\n' && *(data-1) != '>')
496 ++data;
497 break;
498 default:
499 break;
500 }
502 continue;
503 symbols += Symbol(lineNum, token, input, lexem-begin, data-lexem);
504 }
505 }
506 symbols += Symbol(); // eof symbol
507 return symbols;
508}
509
510void Preprocessor::macroExpand(Symbols *into, Preprocessor *that, const Symbols &toExpand, qsizetype &index,
511 int lineNum, bool one, const QSet<QByteArray> &excludeSymbols)
512{
513 SymbolStack symbols;
514 symbols.reserve(8);
515 SafeSymbols sf;
516 sf.symbols = toExpand;
517 sf.index = index;
518 sf.excludedSymbols = excludeSymbols;
519 symbols.push(std::move(sf));
520
521 if (toExpand.isEmpty())
522 return;
523
524 for (;;) {
525 QByteArray macro;
526 Symbols newSyms = macroExpandIdentifier(that, symbols, lineNum, &macro);
527
528 if (macro.isEmpty()) {
529 // not a macro
530 Symbol s = symbols.symbol();
531 s.lineNum = lineNum;
532 *into += s;
533 } else {
534 SafeSymbols sf;
535 sf.symbols = newSyms;
536 sf.index = 0;
537 sf.expandedMacro = macro;
538 symbols.push(std::move(sf));
539 }
540 if (!symbols.hasNext() || (one && symbols.size() == 1))
541 break;
542 symbols.next();
543 }
544
545 if (symbols.size())
546 index = symbols.top().index;
547 else
548 index = toExpand.size();
549}
550
551
552Symbols Preprocessor::macroExpandIdentifier(Preprocessor *that, SymbolStack &symbols, int lineNum, QByteArray *macroName)
553{
554 Symbol s = symbols.symbol();
555
556 // not a macro
557 if (s.token != PP_IDENTIFIER || !that->macros.contains(s) || symbols.dontReplaceSymbol(s.lexem())) {
558 return Symbols();
559 }
560
561 const Macro &macro = that->macros.value(s);
562 *macroName = s.lexem();
563
564 Symbols expansion;
565 if (!macro.isFunction) {
566 expansion = macro.symbols;
567 } else {
568 bool haveSpace = false;
569 while (symbols.test(PP_WHITESPACE)) { haveSpace = true; }
570 if (!symbols.test(PP_LPAREN)) {
571 *macroName = QByteArray();
572 Symbols syms;
573 if (haveSpace)
574 syms += Symbol(lineNum, PP_WHITESPACE);
575 syms += s;
576 syms.last().lineNum = lineNum;
577 return syms;
578 }
579 QVarLengthArray<Symbols, 5> arguments;
580 while (symbols.hasNext()) {
581 Symbols argument;
582 // strip leading space
583 while (symbols.test(PP_WHITESPACE)) {}
584 int nesting = 0;
585 bool vararg = macro.isVariadic && (arguments.size() == macro.arguments.size() - 1);
586 while (symbols.hasNext()) {
587 Token t = symbols.next();
588 if (t == PP_LPAREN) {
589 ++nesting;
590 } else if (t == PP_RPAREN) {
591 --nesting;
592 if (nesting < 0)
593 break;
594 } else if (t == PP_COMMA && nesting == 0) {
595 if (!vararg)
596 break;
597 }
598 argument += symbols.symbol();
599 }
600 arguments += argument;
601
602 if (nesting < 0)
603 break;
604 else if (!symbols.hasNext())
605 that->error("missing ')' in macro usage");
606 }
607
608 // empty VA_ARGS
609 if (macro.isVariadic && arguments.size() == macro.arguments.size() - 1)
610 arguments += Symbols();
611
612 // now replace the macro arguments with the expanded arguments
613 enum Mode {
614 Normal,
615 Hash,
616 HashHash
617 } mode = Normal;
618
619 const auto end = macro.symbols.cend();
620 auto it = macro.symbols.cbegin();
621 const auto lastSym = std::prev(macro.symbols.cend(), !macro.symbols.isEmpty() ? 1 : 0);
622 for (; it != end; ++it) {
623 const Symbol &s = *it;
624 if (s.token == HASH || s.token == PP_HASHHASH) {
625 mode = (s.token == HASH ? Hash : HashHash);
626 continue;
627 }
628 const qsizetype index = macro.arguments.indexOf(s);
629 if (mode == Normal) {
630 if (index >= 0 && index < arguments.size()) {
631 // each argument undoergoes macro expansion if it's not used as part of a # or ##
632 if (it == lastSym || std::next(it)->token != PP_HASHHASH) {
633 Symbols arg = arguments.at(index);
634 qsizetype idx = 1;
635 macroExpand(&expansion, that, arg, idx, lineNum, false, symbols.excludeSymbols());
636 } else {
637 expansion += arguments.at(index);
638 }
639 } else {
640 expansion += s;
641 }
642 } else if (mode == Hash) {
643 if (index < 0) {
644 that->error("'#' is not followed by a macro parameter");
645 continue;
646 } else if (index >= arguments.size()) {
647 that->error("Macro invoked with too few parameters for a use of '#'");
648 continue;
649 }
650
651 const Symbols &arg = arguments.at(index);
652 QByteArray stringified;
653 if (!arg.empty()) {
654 stringified = arg.front().lexem();
655 for (auto it = arg.cbegin(); std::next(it) != arg.cend(); ++it) {
656 const auto next = std::next(it);
657 if (next->from - (it->from + it->len) > 0)
658 stringified += ' ' + next->lexem();
659 else
660 stringified += next->lexem();
661 }
662
663 stringified.replace('"', "\\\"");
664 stringified.prepend('"');
665 stringified.append('"');
666 }
667
668 expansion += Symbol(lineNum, STRING_LITERAL, stringified);
669 } else if (mode == HashHash){
670 if (s.token == WHITESPACE)
671 continue;
672
673 while (expansion.size() && expansion.constLast().token == PP_WHITESPACE)
674 expansion.pop_back();
675
676 Symbol next = s;
677 if (index >= 0 && index < arguments.size()) {
678 const Symbols &arg = arguments.at(index);
679 if (arg.size() == 0) {
680 mode = Normal;
681 continue;
682 }
683 next = arg.at(0);
684 }
685
686 if (!expansion.isEmpty() && expansion.constLast().token == s.token
687 && expansion.constLast().token != STRING_LITERAL) {
688 Symbol last = expansion.takeLast();
689
690 QByteArray lexem = last.lexem() + next.lexem();
691 expansion += Symbol(lineNum, last.token, lexem);
692 } else {
693 expansion += next;
694 }
695
696 if (index >= 0 && index < arguments.size()) {
697 const Symbols &arg = arguments.at(index);
698 if (!arg.isEmpty())
699 expansion.append(arg.cbegin() + 1, arg.cend());
700 }
701 }
702 mode = Normal;
703 }
704 if (mode != Normal)
705 that->error("'#' or '##' found at the end of a macro argument");
706
707 }
708
709 return expansion;
710}
711
713{
714 while (hasNext()) {
715 Token token = next();
716 if (token == PP_IDENTIFIER) {
717 macroExpand(&substituted, this, symbols, index, symbol().lineNum, true);
718 } else if (token == PP_DEFINED) {
719 bool braces = test(PP_LPAREN);
720 if (test(PP_HAS_INCLUDE) || test(PP_HAS_INCLUDE_NEXT)) {
721 // __has_include / __has_include_next are always supported
722 Symbol definedOrNotDefined = symbol();
723 definedOrNotDefined.token = PP_MOC_TRUE;
724 substituted += definedOrNotDefined;
725 } else {
726 next(PP_IDENTIFIER);
727 Symbol definedOrNotDefined = symbol();
728 definedOrNotDefined.token = macros.contains(definedOrNotDefined)? PP_MOC_TRUE : PP_MOC_FALSE;
729 substituted += definedOrNotDefined;
730 }
731 if (braces)
732 test(PP_RPAREN);
733 continue;
734 } else if (token == PP_NEWLINE) {
735 substituted += symbol();
736 break;
737 } else if (token == PP_HAS_INCLUDE || token == PP_HAS_INCLUDE_NEXT) {
738 const bool isIncludeNext = (token == PP_HAS_INCLUDE_NEXT);
739 const Symbol hasIncludeSymbol = symbol();
740 next(LPAREN);
741
742 // Collect the argument, that is everything up to the matching ')'.
743 Symbols argument;
744 int nesting = 0;
745 for (;;) {
746 const Token tok = next();
747 if (tok == PP_RPAREN && nesting == 0)
748 break;
749 if (tok == PP_NEWLINE || tok == NOTOKEN) {
750 const QByteArray msg = "missing ')' in " + hasIncludeSymbol.lexemView();
751 error(hasIncludeSymbol, msg.constData());
752 }
753 if (tok == PP_LPAREN)
754 ++nesting;
755 else if (tok == PP_RPAREN)
756 --nesting;
757 argument += symbol();
758 }
759
760 // an argument that already is a header-name is taken as it stands,
761 // anything else is macro expanded once, and the result then has to
762 // form a header-name.
763 const auto isHeaderName = [](const Symbols &syms) {
764 if (syms.size() == 1)
765 return syms.constFirst().token == PP_STRING_LITERAL;
766 return syms.size() > 2 && syms.constFirst().token == PP_LANGLE
767 && syms.constLast().token == PP_RANGLE;
768 };
769
770 if (!isHeaderName(argument)) {
771 Symbols expanded;
772 qsizetype pos = 1;
773 macroExpand(&expanded, this, argument, pos, hasIncludeSymbol.lineNum, false);
774 // A macro body may carry whitespace, a header-name may not.
775 expanded.removeIf([](const Symbol &s) { return s.token == PP_WHITESPACE; });
776 argument = expanded;
777 if (!isHeaderName(argument)) {
778 const QByteArray msg = "Invalid argument to " + hasIncludeSymbol.lexemView();
779 error(hasIncludeSymbol, msg.constData());
780 }
781 }
782
783 const bool usesAngleInclude = argument.constFirst().token == PP_LANGLE;
784 QByteArray includeAsString;
785 if (usesAngleInclude) {
786 for (qsizetype i = 1; i < argument.size() - 1; ++i)
787 includeAsString += argument.at(i).lexem();
788 } else {
789 includeAsString = argument.constFirst().unquotedLexem();
790 }
791
792 bool result;
793 if (isIncludeNext) {
794 result = !resolveIncludeNext(includeAsString, includeNextStartIndex()).isNull();
795 } else {
796 const QByteArray relative = usesAngleInclude ? QByteArray() : currentFilenames.top();
797 result = !resolveInclude(includeAsString, relative).isNull();
798 }
799 Symbol definedOrNotDefined = hasIncludeSymbol;
800 definedOrNotDefined.token = result ? PP_MOC_TRUE : PP_MOC_FALSE;
801 substituted += definedOrNotDefined;
802 } else {
803 substituted += symbol();
804 }
805 }
806}
807
808
830
832{
833 int value = logical_OR_expression();
834 if (test(PP_QUESTION)) {
835 int alt1 = conditional_expression();
836 int alt2 = test(PP_COLON) ? conditional_expression() : 0;
837 return value ? alt1 : alt2;
838 }
839 return value;
840}
841
843{
844 int value = logical_AND_expression();
845 if (test(PP_OROR))
846 return logical_OR_expression() || value;
847 return value;
848}
849
851{
852 int value = inclusive_OR_expression();
853 if (test(PP_ANDAND))
854 return logical_AND_expression() && value;
855 return value;
856}
857
859{
860 int value = exclusive_OR_expression();
861 if (test(PP_OR))
862 return value | inclusive_OR_expression();
863 return value;
864}
865
867{
868 int value = AND_expression();
869 if (test(PP_HAT))
870 return value ^ exclusive_OR_expression();
871 return value;
872}
873
875{
876 int value = equality_expression();
877 if (test(PP_AND))
878 return value & AND_expression();
879 return value;
880}
881
883{
884 int value = relational_expression();
885 switch (next()) {
886 case PP_EQEQ:
887 return value == equality_expression();
888 case PP_NE:
889 return value != equality_expression();
890 default:
891 prev();
892 return value;
893 }
894}
895
897{
898 int value = shift_expression();
899 switch (next()) {
900 case PP_LANGLE:
901 return value < relational_expression();
902 case PP_RANGLE:
903 return value > relational_expression();
904 case PP_LE:
905 return value <= relational_expression();
906 case PP_GE:
907 return value >= relational_expression();
908 default:
909 prev();
910 return value;
911 }
912}
913
915{
916 int value = additive_expression();
917 switch (next()) {
918 case PP_LTLT:
919 return value << shift_expression();
920 case PP_GTGT:
921 return value >> shift_expression();
922 default:
923 prev();
924 return value;
925 }
926}
927
929{
930 int value = multiplicative_expression();
931 switch (next()) {
932 case PP_PLUS:
933 return value + additive_expression();
934 case PP_MINUS:
935 return value - additive_expression();
936 default:
937 prev();
938 return value;
939 }
940}
941
943{
944 int value = unary_expression();
945 switch (next()) {
946 case PP_STAR:
947 {
948 // get well behaved overflow behavior by converting to long
949 // and then back to int
950 // NOTE: A conformant preprocessor would need to work intmax_t/
951 // uintmax_t according to [cpp.cond], 19.1 §10
952 // But we're not compliant anyway
953 qint64 result = qint64(value) * qint64(multiplicative_expression());
954 return int(result);
955 }
956 case PP_PERCENT:
957 {
958 int remainder = multiplicative_expression();
959 return remainder ? value % remainder : 0;
960 }
961 case PP_SLASH:
962 {
964 return div ? value / div : 0;
965 }
966 default:
967 prev();
968 return value;
969 };
970}
971
973{
974 switch (next()) {
975 case PP_PLUS:
976 return unary_expression();
977 case PP_MINUS:
978 return -unary_expression();
979 case PP_NOT:
980 return !unary_expression();
981 case PP_TILDE:
982 return ~unary_expression();
983 case PP_MOC_TRUE:
984 return 1;
985 case PP_MOC_FALSE:
986 return 0;
987 default:
988 prev();
990 }
991}
992
994{
995 Token t = lookup();
997 || t == PP_PLUS
998 || t == PP_MINUS
999 || t == PP_NOT
1000 || t == PP_TILDE
1001 || t == PP_DEFINED);
1002}
1003
1005{
1006 int value;
1007 if (test(PP_LPAREN)) {
1009 test(PP_RPAREN);
1010 } else {
1011 next();
1012 auto lexView = lexemView();
1013 if (lexView.endsWith('L'))
1014 lexView.chop(1);
1015 value = lexView.toInt(nullptr, 0);
1016 }
1017 return value;
1018}
1019
1021{
1022 Token t = lookup();
1023 return (t == PP_IDENTIFIER
1024 || t == PP_INTEGER_LITERAL
1025 || t == PP_FLOATING_LITERAL
1026 || t == PP_MOC_TRUE
1027 || t == PP_MOC_FALSE
1028 || t == PP_LPAREN);
1029}
1030
1032{
1033 PP_Expression expression;
1034 expression.currentFilenames = currentFilenames;
1035
1036 substituteUntilNewline(expression.symbols);
1037
1038 return expression.value();
1039}
1040
1041static QByteArray readOrMapFile(QFile *file)
1042{
1043 const qint64 size = file->size();
1044 char *rawInput = reinterpret_cast<char*>(file->map(0, size));
1045 return rawInput ? QByteArray::fromRawData(rawInput, size) : file->readAll();
1046}
1047
1049{
1050 Q_ASSERT(len >= 2); // at least `""`
1051 Q_ASSERT(from + len <= lex.size());
1052 Q_ASSERT(next.len >= 2); // at least `""`
1053 Q_ASSERT(next.from + next.len <= next.lex.size());
1054
1055 if (len != lex.size()) {
1056 // "rubbish" around lexem() in `lex`: clean up (`lex` may be the whole file)
1057 QByteArray l = lexemView().chopped(1) % next.lexemView().sliced(1);
1058 lex = std::move(l); // lexemView() aliases `lex`; only clobber it now
1059 from = 0;
1060 } else {
1061 // like QByteArray::append(), but dealing with the "" around each lexem:
1062 const auto unquoted = next.unquotedLexemView();
1063 lex.insert(from + len - 1, // before closing `"`
1064 unquoted);
1065 }
1066 len = lex.size();
1067}
1068
1069static void mergeStringLiterals(Symbols &symbols)
1070{
1071 // like std::unique, but merges instead of skips adjacent STRING_LITERALs:
1072
1073 const auto mergeable = [](const Symbol &lhs, const Symbol &rhs) {
1074 return lhs.token == STRING_LITERAL && rhs.token == STRING_LITERAL;
1075 };
1076
1077 auto end = symbols.end();
1078 auto it = std::adjacent_find(symbols.begin(), symbols.end(), mergeable);
1079 if (it == end) // none found
1080 return;
1081
1082 // we know `it`, `it + 1` are both STRING_LITERAL (adjacent_find post-condition)
1083 // in particular: it + 1 < end
1084
1085 auto dst = it;
1086 auto lit = dst;
1087 ++it;
1088 lit->mergeStringLiteral(*it);
1089
1090 while (++it != end) {
1091 // Loop Invariants:
1092 // - [begin(), dst] is already processed
1093 // - `lit` is the last string literal
1094 // - we can merge if lit == dst
1095 // - [it, end[ still to be checked
1096 if (it->token == STRING_LITERAL) {
1097 if (lit == dst) { // can merge
1098 lit->mergeStringLiteral(*it);
1099 } else { // can't merge: not adjacent to previous STRING_LITERAL
1100 *++dst = std::move(*it);
1101 lit = dst; // remember that this was a literal
1102 }
1103 } else {
1104 *++dst = std::move(*it);
1105 }
1106 }
1107
1108 ++dst;
1109
1110 symbols.erase(dst, end);
1111}
1112
1113// Searches the include path list for \a include, starting at \a startIndex.
1114// For a plain #include startIndex is 0 (search the whole list); for
1115// #include_next it is one past the directory the current file was found in.
1116static IncludeResolution searchIncludePaths(const QList<Parser::IncludePath> &includepaths,
1117 const QByteArray &include,
1118 qsizetype startIndex,
1119 const bool debugIncludes)
1120{
1121 QFileInfo fi;
1122 qsizetype foundIndex = -1;
1123
1124 if (Q_UNLIKELY(debugIncludes)) {
1125 fprintf(stderr, "debug-includes: searching for '%s'\n", include.constData());
1126 }
1127
1128 for (qsizetype i = startIndex; i < includepaths.size(); ++i) {
1129 const Parser::IncludePath &p = includepaths.at(i);
1130 if (fi.exists())
1131 break;
1132
1133 if (p.isFrameworkPath) {
1134 // A framework include names the framework, as in <Framework/Header>.
1135 // An include without a slash is never resolved via a framework
1136 // search path, matching Clang, so that an umbrella include such as
1137 // <QtGui> only resolves through the module's own include directory.
1138 const qsizetype slashPos = include.indexOf('/');
1139 if (slashPos < 0)
1140 continue;
1141 fi.setFile(QString::fromLocal8Bit(p.path + '/' + include.left(slashPos) + ".framework/Headers/"),
1142 QString::fromLocal8Bit(include.mid(slashPos + 1)));
1143 } else {
1144 fi.setFile(QString::fromLocal8Bit(p.path), QString::fromLocal8Bit(include));
1145 }
1146
1147 if (Q_UNLIKELY(debugIncludes)) {
1148 const auto candidate = fi.filePath().toLocal8Bit();
1149 fprintf(stderr, "debug-includes: considering '%s'\n", candidate.constData());
1150 }
1151
1152 // try again, maybe there's a file later in the include paths with the same name
1153 // (186067)
1154 if (fi.isDir()) {
1155 fi = QFileInfo();
1156 continue;
1157 }
1158 foundIndex = i;
1159 }
1160
1161 if (!fi.exists() || fi.isDir()) {
1162 if (Q_UNLIKELY(debugIncludes)) {
1163 fprintf(stderr, "debug-includes: can't find '%s'\n", include.constData());
1164 }
1165 return {};
1166 }
1167
1168 const auto result = fi.canonicalFilePath().toLocal8Bit();
1169
1170 if (Q_UNLIKELY(debugIncludes)) {
1171 fprintf(stderr, "debug-includes: found '%s'\n", result.constData());
1172 }
1173
1174 return { result, foundIndex };
1175}
1176
1177QByteArray Preprocessor::resolveInclude(const QByteArray &include, const QByteArray &relativeTo,
1178 qsizetype *foundIndex)
1179{
1180 if (foundIndex)
1181 *foundIndex = -1;
1182
1183 if (!relativeTo.isEmpty()) {
1184 QFileInfo fi;
1185 fi.setFile(QFileInfo(QString::fromLocal8Bit(relativeTo)).dir(), QString::fromLocal8Bit(include));
1186 if (fi.exists() && !fi.isDir()) {
1187 // Found next to the including file, not via the search path. Inherit
1188 // the includer's include-path position so a later #include_next
1189 // resumes after that directory instead of being mistaken for one in
1190 // the primary source file (e.g. GCC's syslimits.h via limits.h).
1191 if (foundIndex && !currentIncludeDirIndex.empty())
1192 *foundIndex = currentIncludeDirIndex.top();
1193 return fi.canonicalFilePath().toLocal8Bit();
1194 }
1195 }
1196
1197 auto it = nonlocalIncludePathResolutionCache.find(include);
1198 if (it == nonlocalIncludePathResolutionCache.end())
1199 it = nonlocalIncludePathResolutionCache.insert(include,
1200 searchIncludePaths(
1201 includes,
1202 include,
1203 0,
1204 debugIncludes));
1205 if (foundIndex)
1206 *foundIndex = it.value().foundIndex;
1207 return it.value().path;
1208}
1209
1210QByteArray Preprocessor::resolveIncludeNext(const QByteArray &include, qsizetype startIndex,
1211 qsizetype *foundIndex)
1212{
1213 // #include_next never resolves relative to the including file, and its
1214 // result depends on the start index, so it bypasses the resolution cache.
1215 const IncludeResolution resolution = searchIncludePaths(includes, include, startIndex, debugIncludes);
1216 if (foundIndex)
1217 *foundIndex = resolution.foundIndex;
1218 return resolution.path;
1219}
1220
1221qsizetype Preprocessor::includeNextStartIndex()
1222{
1223 // #include_next continues the search after the directory the current file
1224 // was found in. If the current file was not found via the include path
1225 // (the primary source file, or a file included relative to it), search from
1226 // the start of the path, matching GCC/Clang. This is not worth warning
1227 // about: the same file is also seen by the compiler, either as its primary
1228 // source file or, for a moc'ed header, through the relative #include in the
1229 // generated moc_*.cpp, and it resolves the #include_next the same way.
1230 const qsizetype currentDirIndex = currentIncludeDirIndex.top();
1231 if (currentDirIndex < 0) {
1232 if (Q_UNLIKELY(debugIncludes)) {
1233 fprintf(stderr, "debug-includes: '%s' was not found via the include path; "
1234 "#include_next will search from the start\n",
1235 currentFilenames.top().constData());
1236 }
1237 return 0;
1238 }
1239 return currentDirIndex + 1;
1240}
1241
1242void Preprocessor::preprocess(const QByteArray &filename, Symbols &preprocessed, qsizetype includeDirIndex)
1243{
1244 currentFilenames.push(filename);
1245 currentIncludeDirIndex.push(includeDirIndex);
1246 preprocessed.reserve(preprocessed.size() + symbols.size());
1247 while (hasNext()) {
1248 Token token = next();
1249
1250 switch (token) {
1251 case PP_INCLUDE:
1252 case PP_INCLUDE_NEXT:
1253 {
1254 const bool includeNext = (token == PP_INCLUDE_NEXT);
1255 int lineNum = symbol().lineNum;
1256 QByteArray include;
1257 bool local = false;
1258 if (test(PP_STRING_LITERAL)) {
1259 local = lexemView().startsWith('\"');
1260 include = unquotedLexem();
1261 } else {
1262 continue;
1263 }
1264
1265 qsizetype foundIndex = -1;
1266 if (includeNext)
1267 include = resolveIncludeNext(include, includeNextStartIndex(), &foundIndex);
1268 else
1269 include = resolveInclude(include, local ? filename : QByteArray(), &foundIndex);
1270
1271 until(PP_NEWLINE);
1272
1273 if (include.isNull())
1274 continue;
1275
1276 if (Preprocessor::preprocessedIncludes.contains(include))
1277 continue;
1278 Preprocessor::preprocessedIncludes.insert(include);
1279
1280 QFile file(QString::fromLocal8Bit(include.constData()));
1281 if (!file.open(QFile::ReadOnly))
1282 continue;
1283
1284 QByteArray input = readOrMapFile(&file);
1285
1286 file.close();
1287 if (input.isEmpty())
1288 continue;
1289
1290 Symbols saveSymbols = symbols;
1291 qsizetype saveIndex = index;
1292
1293 // phase 1: get rid of backslash-newlines
1294 input = cleaned(input);
1295
1296 // phase 2: tokenize for the preprocessor
1297 symbols = tokenize(input);
1298 input.clear();
1299
1300 index = 0;
1301
1302 // phase 3: preprocess conditions and substitute macros
1303 preprocessed += Symbol(0, MOC_INCLUDE_BEGIN, include);
1304 preprocess(include, preprocessed, foundIndex);
1305 preprocessed += Symbol(lineNum, MOC_INCLUDE_END, include);
1306
1307 symbols = saveSymbols;
1308 index = saveIndex;
1309 continue;
1310 }
1311 case PP_DEFINE:
1312 {
1313 next();
1314 QByteArray name = lexem();
1315 if (name.isEmpty() || !is_ident_start(name[0]))
1316 error();
1317 Macro macro;
1318 macro.isVariadic = false;
1319 if (test(LPAREN)) {
1320 // we have a function macro
1321 macro.isFunction = true;
1323 } else {
1324 macro.isFunction = false;
1325 }
1326 qsizetype start = index;
1327 until(PP_NEWLINE);
1328 macro.symbols.reserve(index - start - 1);
1329
1330 // remove whitespace where there shouldn't be any:
1331 // Before and after the macro, after a # and around ##
1332 Token lastToken = HASH; // skip shitespace at the beginning
1333 for (qsizetype i = start; i < index - 1; ++i) {
1334 Token token = symbols.at(i).token;
1335 if (token == WHITESPACE) {
1336 if (lastToken == PP_HASH || lastToken == HASH ||
1337 lastToken == PP_HASHHASH ||
1338 lastToken == WHITESPACE)
1339 continue;
1340 } else if (token == PP_HASHHASH) {
1341 if (!macro.symbols.isEmpty() &&
1342 lastToken == WHITESPACE)
1343 macro.symbols.pop_back();
1344 }
1345 macro.symbols.append(symbols.at(i));
1346 lastToken = token;
1347 }
1348 // remove trailing whitespace
1349 while (!macro.symbols.isEmpty() &&
1350 (macro.symbols.constLast().token == PP_WHITESPACE || macro.symbols.constLast().token == WHITESPACE))
1351 macro.symbols.pop_back();
1352
1353 if (!macro.symbols.isEmpty()) {
1354 if (macro.symbols.constFirst().token == PP_HASHHASH ||
1355 macro.symbols.constLast().token == PP_HASHHASH) {
1356 error("'##' cannot appear at either end of a macro expansion");
1357 }
1358 }
1359 macros.insert(name, macro);
1360 continue;
1361 }
1362 case PP_UNDEF: {
1363 next();
1364 QByteArray name = lexem();
1365 until(PP_NEWLINE);
1366 macros.remove(name);
1367 continue;
1368 }
1369 case PP_IDENTIFIER: {
1370 // substitute macros
1371 macroExpand(&preprocessed, this, symbols, index, symbol().lineNum, true);
1372 continue;
1373 }
1374 case PP_HASH:
1375 until(PP_NEWLINE);
1376 continue; // skip unknown preprocessor statement
1377 case PP_IFDEF:
1378 case PP_IFNDEF:
1379 case PP_IF:
1380 while (!evaluateCondition()) {
1381 if (!skipBranch())
1382 break;
1383 if (test(PP_ELIF)) {
1384 } else {
1385 until(PP_NEWLINE);
1386 break;
1387 }
1388 }
1389 continue;
1390 case PP_ELIF:
1391 case PP_ELSE:
1393 Q_FALLTHROUGH();
1394 case PP_ENDIF:
1395 until(PP_NEWLINE);
1396 continue;
1397 case PP_NEWLINE:
1398 continue;
1399 case SIGNALS:
1400 case SLOTS: {
1401 Symbol sym = symbol();
1402 if (macros.contains("QT_NO_KEYWORDS"))
1403 sym.token = IDENTIFIER;
1404 else
1405 sym.token = (token == SIGNALS ? Q_SIGNALS_TOKEN : Q_SLOTS_TOKEN);
1406 preprocessed += sym;
1407 } continue;
1408 default:
1409 break;
1410 }
1411 preprocessed += symbol();
1412 }
1413
1414 currentIncludeDirIndex.pop();
1415 currentFilenames.pop();
1416}
1417
1418Symbols Preprocessor::preprocessed(const QByteArray &filename, QFile *file)
1419{
1420 QByteArray input = readOrMapFile(file);
1421
1422 if (input.isEmpty())
1423 return symbols;
1424
1425 // phase 1: get rid of backslash-newlines
1426 input = cleaned(input);
1427
1428 // phase 2: tokenize for the preprocessor
1429 index = 0;
1430 symbols = tokenize(input);
1431
1432#if 0
1433 for (int j = 0; j < symbols.size(); ++j)
1434 fprintf(stderr, "line %d: %s(%s)\n",
1435 symbols[j].lineNum,
1436 symbols[j].lexem().constData(),
1437 tokenTypeName(symbols[j].token));
1438#endif
1439
1440 // phase 3: preprocess conditions and substitute macros
1441 Symbols result;
1442 // Preallocate some space to speed up the code below.
1443 // The magic value was found by logging the final size
1444 // and calculating an average when running moc over FOSS projects.
1445 result.reserve(file->size() / 300000);
1446 preprocess(filename, result);
1447 mergeStringLiterals(result);
1448
1449#if 0
1450 for (int j = 0; j < result.size(); ++j)
1451 fprintf(stderr, "line %d: %s(%s)\n",
1452 result[j].lineNum,
1453 result[j].lexem().constData(),
1454 tokenTypeName(result[j].token));
1455#endif
1456
1457 return result;
1458}
1459
1461{
1462 Symbols arguments;
1463 while (hasNext()) {
1464 while (test(PP_WHITESPACE)) {}
1465 Token t = next();
1466 if (t == PP_RPAREN)
1467 break;
1468 if (t != PP_IDENTIFIER) {
1469 QByteArrayView l = lexemView();
1470 if (l == "...") {
1471 m->isVariadic = true;
1472 arguments += Symbol(symbol().lineNum, PP_IDENTIFIER, "__VA_ARGS__");
1473 while (test(PP_WHITESPACE)) {}
1474 if (!test(PP_RPAREN))
1475 error("missing ')' in macro argument list");
1476 break;
1477 } else if (!is_identifier(l.constData(), l.size())) {
1478 error("Unexpected character in macro argument list.");
1479 }
1480 }
1481
1482 Symbol arg = symbol();
1483 if (arguments.contains(arg))
1484 error("Duplicate macro parameter.");
1485 arguments += symbol();
1486
1487 while (test(PP_WHITESPACE)) {}
1488 t = next();
1489 if (t == PP_RPAREN)
1490 break;
1491 if (t == PP_COMMA)
1492 continue;
1493 if (lexemView() == "...") {
1494 //GCC extension: #define FOO(x, y...) x(y)
1495 // The last argument was already parsed. Just mark the macro as variadic.
1496 m->isVariadic = true;
1497 while (test(PP_WHITESPACE)) {}
1498 if (!test(PP_RPAREN))
1499 error("missing ')' in macro argument list");
1500 break;
1501 }
1502 error("Unexpected character in macro argument list.");
1503 }
1504 m->arguments = arguments;
1505 while (test(PP_WHITESPACE)) {}
1506}
1507
1508void Preprocessor::until(Token t)
1509{
1510 while(hasNext() && next() != t)
1511 ;
1512}
1513
1515{
1516 debugIncludes = value;
1517}
1518
1519
1520QT_END_NAMESPACE
int relational_expression()
int exclusive_OR_expression()
bool unary_expression_lookup()
int logical_OR_expression()
int equality_expression()
int logical_AND_expression()
int additive_expression()
int multiplicative_expression()
int conditional_expression()
bool primary_expression_lookup()
int inclusive_OR_expression()
QByteArray resolveIncludeNext(const QByteArray &filename, qsizetype startIndex, qsizetype *foundIndex=nullptr)
void setDebugIncludes(bool value)
void parseDefineArguments(Macro *m)
void skipUntilEndif()
Symbols preprocessed(const QByteArray &filename, QFile *device)
void substituteUntilNewline(Symbols &substituted)
static bool preprocessOnly
QByteArray resolveInclude(const QByteArray &filename, const QByteArray &relativeTo, qsizetype *foundIndex=nullptr)
@ PreparePreprocessorStatement
@ TokenizePreprocessorStatement
Definition qlist.h:82
const Symbol & symbol() const
Definition symbols.h:102
bool hasNext()
Definition symbols.h:89
Token next()
Definition symbols.h:94
bool test(Token)
Definition symbols.h:111
short defnext
Definition keywords.cpp:455
static const short keyword_trans[][128]
Definition keywords.cpp:7
Token token
Definition keywords.cpp:452
Token ident
Definition keywords.cpp:456
short next
Definition keywords.cpp:453
char defchar
Definition keywords.cpp:454
Combined button and popup list for selecting options.
short next
static const short pp_keyword_trans[][128]
Definition ppkeywords.cpp:7
PP_Token ident
short defnext
PP_Token token
char defchar
static QByteArray readOrMapFile(QFile *file)
static IncludeResolution searchIncludePaths(const QList< Parser::IncludePath > &includepaths, const QByteArray &include, qsizetype startIndex, const bool debugIncludes)
static QByteArray cleaned(const QByteArray &input)
static void mergeStringLiterals(Symbols &symbols)
bool is_ident_char(char s)
Definition utils.h:30
const char * skipQuote(const char *data)
Definition utils.h:42
bool is_space(char s)
Definition utils.h:19
Simple structure used by the Doc and DocParser classes.
bool isVariadic
bool isFunction
Symbol(int lineNum, Token token)
Definition symbols.h:48
Token token
Definition symbols.h:58
void mergeStringLiteral(const Symbol &next)
int lineNum
Definition symbols.h:57
Symbol()=default
QList< Symbol > Symbols
Definition symbols.h:75