clang 24.0.0git
Preprocessor.cpp
Go to the documentation of this file.
1//===- Preprocessor.cpp - C Language Family Preprocessor Implementation ---===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the Preprocessor interface.
10//
11//===----------------------------------------------------------------------===//
12//
13// Options to support:
14// -H - Print the name of each header file used.
15// -d[DNI] - Dump various things.
16// -fworking-directory - #line's with preprocessor's working dir.
17// -fpreprocessed
18// -dependency-file,-M,-MM,-MF,-MG,-MP,-MT,-MQ,-MD,-MMD
19// -W*
20// -w
21//
22// Messages to emit:
23// "Multiple include guards may be useful for:\n"
24//
25//===----------------------------------------------------------------------===//
26
31#include "clang/Basic/LLVM.h"
33#include "clang/Basic/Module.h"
42#include "clang/Lex/Lexer.h"
44#include "clang/Lex/MacroArgs.h"
45#include "clang/Lex/MacroInfo.h"
48#include "clang/Lex/Pragma.h"
53#include "clang/Lex/Token.h"
56#include "llvm/ADT/APInt.h"
57#include "llvm/ADT/ArrayRef.h"
58#include "llvm/ADT/DenseMap.h"
59#include "llvm/ADT/STLExtras.h"
60#include "llvm/ADT/ScopeExit.h"
61#include "llvm/ADT/SmallVector.h"
62#include "llvm/ADT/StringRef.h"
63#include "llvm/Support/Capacity.h"
64#include "llvm/Support/ErrorHandling.h"
65#include "llvm/Support/FormatVariadic.h"
66#include "llvm/Support/MemoryBuffer.h"
67#include "llvm/Support/MemoryBufferRef.h"
68#include "llvm/Support/SaveAndRestore.h"
69#include "llvm/Support/raw_ostream.h"
70#include <algorithm>
71#include <cassert>
72#include <memory>
73#include <optional>
74#include <string>
75#include <utility>
76#include <vector>
77
78using namespace clang;
79
80/// Minimum distance between two check points, in tokens.
81static constexpr unsigned CheckPointStepSize = 1024;
82
84
86
88 DiagnosticsEngine &diags, const LangOptions &opts,
89 SourceManager &SM, HeaderSearch &Headers,
90 ModuleLoader &TheModuleLoader,
91 IdentifierInfoLookup *IILookup, bool OwnsHeaders,
93 : PPOpts(PPOpts), Diags(&diags), LangOpts(opts),
94 FileMgr(Headers.getFileMgr()), SourceMgr(SM),
95 ScratchBuf(new ScratchBuffer(SourceMgr)), HeaderInfo(Headers),
96 TheModuleLoader(TheModuleLoader), ExternalSource(nullptr),
97 // As the language options may have not been loaded yet (when
98 // deserializing an ASTUnit), adding keywords to the identifier table is
99 // deferred to Preprocessor::Initialize().
100 Identifiers(IILookup), PragmaHandlers(new PragmaNamespace(StringRef())),
101 TUKind(TUKind), SkipMainFilePreamble(0, true),
102 CurSubmoduleState(&NullSubmoduleState) {
103 OwnsHeaderSearch = OwnsHeaders;
104
105 // Only record check points if we might highlight diagnostic snippets.
106 RecordCheckPoints = getDiagnostics().getShowColors();
107
108 // Default to discarding comments.
109 KeepComments = false;
110 KeepMacroComments = false;
111 SuppressIncludeNotFoundError = false;
112
113 // Macro expansion is enabled.
114 DisableMacroExpansion = false;
115 MacroExpansionInDirectivesOverride = false;
116 InMacroArgs = false;
117 ArgMacro = nullptr;
118 InMacroArgPreExpansion = false;
119 NumCachedTokenLexers = 0;
120 PragmasEnabled = true;
121 ParsingIfOrElifDirective = false;
122 PreprocessedOutput = false;
123
124 // We haven't read anything from the external source.
125 ReadMacrosFromExternalSource = false;
126
127 LastExportKeyword.startToken();
128
129 BuiltinInfo = std::make_unique<Builtin::Context>();
130
131 // "Poison" __VA_ARGS__, __VA_OPT__ which can only appear in the expansion of
132 // a macro. They get unpoisoned where it is allowed.
133 (Ident__VA_ARGS__ = getIdentifierInfo("__VA_ARGS__"))->setIsPoisoned();
134 SetPoisonReason(Ident__VA_ARGS__,diag::ext_pp_bad_vaargs_use);
135 (Ident__VA_OPT__ = getIdentifierInfo("__VA_OPT__"))->setIsPoisoned();
136 SetPoisonReason(Ident__VA_OPT__,diag::ext_pp_bad_vaopt_use);
137
138 // Initialize the pragma handlers.
139 RegisterBuiltinPragmas();
140
141 // Initialize builtin macros like __LINE__ and friends.
142 RegisterBuiltinMacros();
143
144 if(LangOpts.Borland) {
145 Ident__exception_info = getIdentifierInfo("_exception_info");
146 Ident___exception_info = getIdentifierInfo("__exception_info");
147 Ident_GetExceptionInfo = getIdentifierInfo("GetExceptionInformation");
148 Ident__exception_code = getIdentifierInfo("_exception_code");
149 Ident___exception_code = getIdentifierInfo("__exception_code");
150 Ident_GetExceptionCode = getIdentifierInfo("GetExceptionCode");
151 Ident__abnormal_termination = getIdentifierInfo("_abnormal_termination");
152 Ident___abnormal_termination = getIdentifierInfo("__abnormal_termination");
153 Ident_AbnormalTermination = getIdentifierInfo("AbnormalTermination");
154 } else {
155 Ident__exception_info = Ident__exception_code = nullptr;
156 Ident__abnormal_termination = Ident___exception_info = nullptr;
157 Ident___exception_code = Ident___abnormal_termination = nullptr;
158 Ident_GetExceptionInfo = Ident_GetExceptionCode = nullptr;
159 Ident_AbnormalTermination = nullptr;
160 }
161
162 Ident__GLIBCXX__ = getIdentifierInfo("__GLIBCXX__");
163
164 // Default incremental processing to -fincremental-extensions, clients can
165 // override with `enableIncrementalProcessing` if desired.
166 IncrementalProcessing = LangOpts.IncrementalExtensions;
167
168 // If using a PCH where a #pragma hdrstop is expected, start skipping tokens.
170 SkippingUntilPragmaHdrStop = true;
171
172 // If using a PCH with a through header, start skipping tokens.
173 if (!this->PPOpts.PCHThroughHeader.empty() &&
174 !this->PPOpts.ImplicitPCHInclude.empty())
175 SkippingUntilPCHThroughHeader = true;
176
177 if (this->PPOpts.GeneratePreamble)
178 PreambleConditionalStack.startRecording();
179
180 MaxTokens = LangOpts.MaxTokens;
181}
182
184 assert(!isBacktrackEnabled() && "EnableBacktrack/Backtrack imbalance!");
185
186 IncludeMacroStack.clear();
187
188 // Free any cached macro expanders.
189 // This populates MacroArgCache, so all TokenLexers need to be destroyed
190 // before the code below that frees up the MacroArgCache list.
191 std::fill(TokenLexerCache, TokenLexerCache + NumCachedTokenLexers, nullptr);
192 CurTokenLexer.reset();
193
194 // Free any cached MacroArgs.
195 for (MacroArgs *ArgList = MacroArgCache; ArgList;)
196 ArgList = ArgList->deallocate();
197
198 // Delete the header search info, if we own it.
199 if (OwnsHeaderSearch)
200 delete &HeaderInfo;
201}
202
204 const TargetInfo *AuxTarget) {
205 assert((!this->Target || this->Target == &Target) &&
206 "Invalid override of target information");
207 this->Target = &Target;
208
209 assert((!this->AuxTarget || this->AuxTarget == AuxTarget) &&
210 "Invalid override of aux target information.");
211 this->AuxTarget = AuxTarget;
212
213 // Initialize information about built-ins.
214 BuiltinInfo->InitializeTarget(Target, AuxTarget);
215 HeaderInfo.setTarget(Target);
216
217 // Populate the identifier table with info about keywords for the current language.
218 Identifiers.AddKeywords(LangOpts);
219
220 // Initialize the __FTL_EVAL_METHOD__ macro to the TargetInfo.
221 setTUFPEvalMethod(getTargetInfo().getFPEvalMethod());
222
223 if (getLangOpts().getFPEvalMethod() == LangOptions::FEM_UnsetOnCommandLine)
224 // Use setting from TargetInfo.
225 setCurrentFPEvalMethod(SourceLocation(), Target.getFPEvalMethod());
226 else
227 // Set initial value of __FLT_EVAL_METHOD__ from the command line.
228 setCurrentFPEvalMethod(SourceLocation(), getLangOpts().getFPEvalMethod());
229}
230
232 NumEnteredSourceFiles = 0;
233
234 // Reset pragmas
235 PragmaHandlersBackup = std::move(PragmaHandlers);
236 PragmaHandlers = std::make_unique<PragmaNamespace>(StringRef());
237 RegisterBuiltinPragmas();
238
239 // Reset PredefinesFileID
240 PredefinesFileID = FileID();
241}
242
244 NumEnteredSourceFiles = 1;
245
246 PragmaHandlers = std::move(PragmaHandlersBackup);
247}
248
249void Preprocessor::DumpToken(const Token &Tok, bool DumpFlags) const {
250 std::string TokenStr;
251 llvm::raw_string_ostream OS(TokenStr);
252
253 // The alignment of 16 is chosen to comfortably fit most identifiers.
254 OS << llvm::formatv("{0,-16} ", tok::getTokenName(Tok.getKind()));
255
256 // Annotation tokens are just markers that don't have a spelling -- they
257 // indicate where something expanded.
258 if (!Tok.isAnnotation()) {
259 OS << "'";
260 // Escape string to prevent token spelling from spanning multiple lines.
261 OS.write_escaped(getSpelling(Tok));
262 OS << "'";
263 }
264
265 // The alignment of 48 (32 characters for the spelling + the 16 for
266 // the identifier name) fits most variable names, keywords and annotations.
267 llvm::errs() << llvm::formatv("{0,-48} ", OS.str());
268
269 if (!DumpFlags) return;
270
271 auto Loc = Tok.getLocation();
272 llvm::errs() << "Loc=<";
273 DumpLocation(Loc);
274 llvm::errs() << ">";
275
276 // If the token points directly to a file location (i.e. not a macro
277 // expansion), then add additional padding so that trailing markers
278 // align, provided the line/column numbers are reasonably sized.
279 //
280 // Otherwise, if it's a macro expansion, don't bother with alignment,
281 // as the line will include multiple locations and be very long.
282 //
283 // NOTE: To keep this stateless, it doesn't account for filename
284 // length, so when a header starts markers will be temporarily misaligned.
285 if (Loc.isFileID()) {
286 PresumedLoc PLoc = SourceMgr.getPresumedLoc(Loc);
287
288 if (!PLoc.isInvalid()) {
289 int LineWidth = llvm::utostr(PLoc.getLine()).size();
290 int ColumnWidth = llvm::utostr(PLoc.getColumn()).size();
291
292 // Reserve space for lines up to 9999 and columns up to 99,
293 // which is 4 + 2 = 6 characters in total.
294 const int ReservedSpace = 6;
295
296 int LeftSpace = ReservedSpace - LineWidth - ColumnWidth;
297 int Padding = std::max<int>(0, LeftSpace);
298
299 llvm::errs().indent(Padding);
300 }
301 }
302
303 if (Tok.isAtStartOfLine())
304 llvm::errs() << " [StartOfLine]";
305 if (Tok.hasLeadingSpace())
306 llvm::errs() << " [LeadingSpace]";
307 if (Tok.isExpandDisabled())
308 llvm::errs() << " [ExpandDisabled]";
309 if (Tok.needsCleaning()) {
310 const char *Start = SourceMgr.getCharacterData(Tok.getLocation());
311 llvm::errs() << " [UnClean='" << StringRef(Start, Tok.getLength()) << "']";
312 }
313}
314
316 Loc.print(llvm::errs(), SourceMgr);
317}
318
319void Preprocessor::DumpMacro(const MacroInfo &MI) const {
320 llvm::errs() << "MACRO: ";
321 for (unsigned i = 0, e = MI.getNumTokens(); i != e; ++i) {
323 llvm::errs() << " ";
324 }
325 llvm::errs() << "\n";
326}
327
329 llvm::errs() << "\n*** Preprocessor Stats:\n";
330 llvm::errs() << NumDirectives << " directives found:\n";
331 llvm::errs() << " " << NumDefined << " #define.\n";
332 llvm::errs() << " " << NumUndefined << " #undef.\n";
333 llvm::errs() << " #include/#include_next/#import:\n";
334 llvm::errs() << " " << NumEnteredSourceFiles << " source files entered.\n";
335 llvm::errs() << " " << MaxIncludeStackDepth << " max include stack depth\n";
336 llvm::errs() << " " << NumIf << " #if/#ifndef/#ifdef.\n";
337 llvm::errs() << " " << NumElse << " #else/#elif/#elifdef/#elifndef.\n";
338 llvm::errs() << " " << NumEndif << " #endif.\n";
339 llvm::errs() << " " << NumPragma << " #pragma.\n";
340 llvm::errs() << NumSkipped << " #if/#ifndef#ifdef regions skipped\n";
341
342 llvm::errs() << NumMacroExpanded << "/" << NumFnMacroExpanded << "/"
343 << NumBuiltinMacroExpanded << " obj/fn/builtin macros expanded, "
344 << NumFastMacroExpanded << " on the fast path.\n";
345 llvm::errs() << (NumFastTokenPaste+NumTokenPaste)
346 << " token paste (##) operations performed, "
347 << NumFastTokenPaste << " on the fast path.\n";
348
349 llvm::errs() << "\nPreprocessor Memory: " << getTotalMemory() << "B total";
350
351 llvm::errs() << "\n BumpPtr: " << BP.getTotalMemory();
352 llvm::errs() << "\n Macro Expanded Tokens: "
353 << llvm::capacity_in_bytes(MacroExpandedTokens);
354 llvm::errs() << "\n Predefines Buffer: " << Predefines.capacity();
355 // FIXME: List information for all submodules.
356 llvm::errs() << "\n Macros: "
357 << llvm::capacity_in_bytes(CurSubmoduleState->Macros);
358 llvm::errs() << "\n #pragma push_macro Info: "
359 << llvm::capacity_in_bytes(PragmaPushMacroInfo);
360 llvm::errs() << "\n Poison Reasons: "
361 << llvm::capacity_in_bytes(PoisonReasons);
362 llvm::errs() << "\n Comment Handlers: "
363 << llvm::capacity_in_bytes(CommentHandlers) << "\n";
364}
365
366llvm::iterator_range<Preprocessor::macro_iterator>
367Preprocessor::macros(bool IncludeExternalMacros) const {
368 if (IncludeExternalMacros && ExternalSource &&
369 !ReadMacrosFromExternalSource) {
370 ReadMacrosFromExternalSource = true;
371 ExternalSource->ReadDefinedMacros();
372 }
373 // Make sure we cover all macros in visible modules.
374 for (const ModuleMacro &Macro : ModuleMacros)
375 CurSubmoduleState->Macros.try_emplace(Macro.II);
376
377 return CurSubmoduleState->Macros;
378}
379
381 return BP.getTotalMemory()
382 + llvm::capacity_in_bytes(MacroExpandedTokens)
383 + Predefines.capacity() /* Predefines buffer. */
384 // FIXME: Include sizes from all submodules, and include MacroInfo sizes,
385 // and ModuleMacros.
386 + llvm::capacity_in_bytes(CurSubmoduleState->Macros)
387 + llvm::capacity_in_bytes(PragmaPushMacroInfo)
388 + llvm::capacity_in_bytes(PoisonReasons)
389 + llvm::capacity_in_bytes(CommentHandlers);
390}
391
392/// Compares macro tokens with a specified token value sequence.
393static bool MacroDefinitionEquals(const MacroInfo *MI,
394 ArrayRef<TokenValue> Tokens) {
395 return Tokens.size() == MI->getNumTokens() &&
396 std::equal(Tokens.begin(), Tokens.end(), MI->tokens_begin());
397}
398
400 SourceLocation Loc,
401 ArrayRef<TokenValue> Tokens) const {
402 SourceLocation BestLocation;
403 StringRef BestSpelling;
404 for (const auto &M : macros()) {
405 const MacroDirective::DefInfo Def =
406 M.second.findDirectiveAtLoc(Loc, SourceMgr);
407 if (!Def || !Def.getMacroInfo())
408 continue;
409 if (!Def.getMacroInfo()->isObjectLike())
410 continue;
411 if (!MacroDefinitionEquals(Def.getMacroInfo(), Tokens))
412 continue;
413 SourceLocation Location = Def.getLocation();
414 // Choose the macro defined latest.
415 if (BestLocation.isInvalid() ||
416 (Location.isValid() &&
417 SourceMgr.isBeforeInTranslationUnit(BestLocation, Location))) {
418 BestLocation = Location;
419 BestSpelling = M.first->getName();
420 }
421 }
422 return BestSpelling;
423}
424
426 if (InCachingLexMode())
427 CurLexerCallback = CLK_CachingLexer;
428 else if (CurLexer)
429 CurLexerCallback = CurLexer->isDependencyDirectivesLexer()
430 ? CLK_DependencyDirectivesLexer
431 : CLK_Lexer;
432 else if (CurTokenLexer)
433 CurLexerCallback = CLK_TokenLexer;
434 else
435 CurLexerCallback = CLK_Lexer;
436}
437
439 unsigned CompleteLine,
440 unsigned CompleteColumn) {
441 assert(CompleteLine && CompleteColumn && "Starts from 1:1");
442 assert(!CodeCompletionFile && "Already set");
443
444 // Load the actual file's contents.
445 std::optional<llvm::MemoryBufferRef> Buffer =
446 SourceMgr.getMemoryBufferForFileOrNone(File);
447 if (!Buffer)
448 return true;
449
450 // Find the byte position of the truncation point.
451 const char *Position = Buffer->getBufferStart();
452 for (unsigned Line = 1; Line < CompleteLine; ++Line) {
453 for (; *Position; ++Position) {
454 if (*Position != '\r' && *Position != '\n')
455 continue;
456
457 // Eat \r\n or \n\r as a single line.
458 if ((Position[1] == '\r' || Position[1] == '\n') &&
459 Position[0] != Position[1])
460 ++Position;
461 ++Position;
462 break;
463 }
464 }
465
466 Position += CompleteColumn - 1;
467
468 // If pointing inside the preamble, adjust the position at the beginning of
469 // the file after the preamble.
470 if (SkipMainFilePreamble.first &&
471 SourceMgr.getFileEntryForID(SourceMgr.getMainFileID()) == File) {
472 if (Position - Buffer->getBufferStart() < SkipMainFilePreamble.first)
473 Position = Buffer->getBufferStart() + SkipMainFilePreamble.first;
474 }
475
476 if (Position > Buffer->getBufferEnd())
477 Position = Buffer->getBufferEnd();
478
479 CodeCompletionFile = File;
480 CodeCompletionOffset = Position - Buffer->getBufferStart();
481
482 auto NewBuffer = llvm::WritableMemoryBuffer::getNewUninitMemBuffer(
483 Buffer->getBufferSize() + 1, Buffer->getBufferIdentifier());
484 char *NewBuf = NewBuffer->getBufferStart();
485 char *NewPos = std::copy(Buffer->getBufferStart(), Position, NewBuf);
486 *NewPos = '\0';
487 std::copy(Position, Buffer->getBufferEnd(), NewPos+1);
488 SourceMgr.overrideFileContents(File, std::move(NewBuffer));
489
490 return false;
491}
492
494 bool IsAngled) {
496 if (CodeComplete)
497 CodeComplete->CodeCompleteIncludedFile(Dir, IsAngled);
498}
499
502 if (CodeComplete)
503 CodeComplete->CodeCompleteNaturalLanguage();
504}
505
506/// getSpelling - This method is used to get the spelling of a token into a
507/// SmallVector. Note that the returned StringRef may not point to the
508/// supplied buffer if a copy can be avoided.
510 SmallVectorImpl<char> &Buffer,
511 bool *Invalid) const {
512 // NOTE: this has to be checked *before* testing for an IdentifierInfo.
513 if (Tok.isNot(tok::raw_identifier) && !Tok.hasUCN()) {
514 // Try the fast path.
515 if (const IdentifierInfo *II = Tok.getIdentifierInfo())
516 return II->getName();
517 }
518
519 // Resize the buffer if we need to copy into it.
520 if (Tok.needsCleaning())
521 Buffer.resize(Tok.getLength());
522
523 const char *Ptr = Buffer.data();
524 unsigned Len = getSpelling(Tok, Ptr, Invalid);
525 return StringRef(Ptr, Len);
526}
527
528/// CreateString - Plop the specified string into a scratch buffer and return a
529/// location for it. If specified, the source location provides a source
530/// location for the token.
532 SourceLocation ExpansionLocStart,
533 SourceLocation ExpansionLocEnd) {
534 Tok.setLength(Str.size());
535
536 const char *DestPtr;
537 SourceLocation Loc = ScratchBuf->getToken(Str.data(), Str.size(), DestPtr);
538
539 if (ExpansionLocStart.isValid())
540 Loc = SourceMgr.createExpansionLoc(Loc, ExpansionLocStart,
541 ExpansionLocEnd, Str.size());
542 Tok.setLocation(Loc);
543
544 // If this is a raw identifier or a literal token, set the pointer data.
545 if (Tok.is(tok::raw_identifier))
546 Tok.setRawIdentifierData(DestPtr);
547 else if (Tok.isLiteral())
548 Tok.setLiteralData(DestPtr);
549}
550
552 auto &SM = getSourceManager();
553 SourceLocation SpellingLoc = SM.getSpellingLoc(Loc);
554 FileIDAndOffset LocInfo = SM.getDecomposedLoc(SpellingLoc);
555 bool Invalid = false;
556 StringRef Buffer = SM.getBufferData(LocInfo.first, &Invalid);
557 if (Invalid)
558 return SourceLocation();
559
560 // FIXME: We could consider re-using spelling for tokens we see repeatedly.
561 const char *DestPtr;
562 SourceLocation Spelling =
563 ScratchBuf->getToken(Buffer.data() + LocInfo.second, Length, DestPtr);
564 return SM.createTokenSplitLoc(Spelling, Loc, Loc.getLocWithOffset(Length));
565}
566
568 if (!getLangOpts().isCompilingModule())
569 return nullptr;
570
571 return getHeaderSearchInfo().lookupModule(getLangOpts().CurrentModule);
572}
573
575 if (!getLangOpts().isCompilingModuleImplementation())
576 return nullptr;
577
578 return getHeaderSearchInfo().lookupModule(getLangOpts().ModuleName);
579}
580
581//===----------------------------------------------------------------------===//
582// Preprocessor Initialization Methods
583//===----------------------------------------------------------------------===//
584
585/// EnterMainSourceFile - Enter the specified FileID as the main source file,
586/// which implicitly adds the builtin defines etc.
588 // We do not allow the preprocessor to reenter the main file. Doing so will
589 // cause FileID's to accumulate information from both runs (e.g. #line
590 // information) and predefined macros aren't guaranteed to be set properly.
591 assert(NumEnteredSourceFiles == 0 && "Cannot reenter the main file!");
592 FileID MainFileID = SourceMgr.getMainFileID();
593
594 // Whether and how the main file starts a C++20 module unit. Implicit inputs
595 // are placed in its existing global module fragment, or in a synthesized one
596 // for a named module without a GMF.
597 ModuleUnitKind MainFileModuleUnitKind = ModuleUnitKind::NotModuleUnit;
598
599 // If MainFileID is loaded it means we loaded an AST file, no need to enter
600 // a main file.
601 if (!SourceMgr.isLoadedFileID(MainFileID)) {
602 // Enter the main file source buffer.
603 EnterSourceFile(MainFileID, nullptr, SourceLocation());
604
605 // If we've been asked to skip bytes in the main file (e.g., as part of a
606 // precompiled preamble), do so now.
607 if (SkipMainFilePreamble.first > 0)
608 CurLexer->SetByteOffset(SkipMainFilePreamble.first,
609 SkipMainFilePreamble.second);
610
611 // Tell the header info that the main file was entered. If the file is later
612 // #imported, it won't be re-entered.
613 if (OptionalFileEntryRef FE = SourceMgr.getFileEntryRefForID(MainFileID))
614 markIncluded(*FE);
615
616 // Record the first PP token in the main file. This is used to generate
617 // better diagnostics for C++ modules.
618 //
619 // // This is a comment.
620 // #define FOO int // note: add 'module;' to the start of the file
621 // ^ FirstPPToken // to introduce a global module fragment.
622 //
623 // export module M; // error: module declaration must occur
624 // // at the start of the translation unit.
625 if (getLangOpts().CPlusPlusModules) {
626 std::optional<StringRef> Input =
628 if (!isPreprocessedModuleFile() && Input)
629 MainFileIsPreprocessedModuleFile =
631 if (Input && !MainFileIsPreprocessedModuleFile && hasDeferredGMFInputs())
632 MainFileModuleUnitKind = scanInputForCXX20ModuleUnit(*Input);
633 auto Tracer = std::make_unique<NoTrivialPPDirectiveTracer>(*this);
634 DirTracer = Tracer.get();
635 addPPCallbacks(std::move(Tracer));
636 std::optional<Token> FirstPPTok = CurLexer->peekNextPPToken();
637 if (FirstPPTok)
638 FirstPPTokenLoc = FirstPPTok->getLocation();
639 }
640 }
641
642 // Preserve the historical placement in the predefines buffer for ordinary
643 // translation units. A module unit opening with `module;` leaves the inputs
644 // deferred until the introducer has been lexed. For a named module without a
645 // GMF, synthesize the introducer before the main file and use the same
646 // deferred-input path.
647 if (hasDeferredGMFInputs()) {
648 if (PredefinesWereReplaced) {
649 // Loading an implicit PCH replaces Predefines with the directives
650 // suggested by ASTReader. For a module unit, those are the only implicit
651 // inputs that still need to be processed in the GMF.
652 if (MainFileModuleUnitKind != ModuleUnitKind::NotModuleUnit) {
653 DeferredGMFInputs = std::move(Predefines);
654 Predefines.clear();
655 }
656 } else if (MainFileModuleUnitKind == ModuleUnitKind::NotModuleUnit) {
657 // Preserve the historical predefines ordering for an ordinary
658 // translation unit.
659 Predefines += DeferredGMFInputs;
660 }
661 if (MainFileModuleUnitKind ==
663 Predefines += "# 1 \"<implicit-global-module-fragment>\" 1\nmodule;\n";
664 HasSynthesizedGMF = true;
665 } else if (MainFileModuleUnitKind == ModuleUnitKind::NotModuleUnit) {
666 DeferredGMFInputs.clear();
667 }
668 }
669
670 // Preprocess Predefines to populate the initial preprocessor state.
671 std::unique_ptr<llvm::MemoryBuffer> SB =
672 llvm::MemoryBuffer::getMemBufferCopy(Predefines, "<built-in>");
673 assert(SB && "Cannot create predefined source buffer");
674 FileID FID = SourceMgr.createFileID(std::move(SB));
675 assert(FID.isValid() && "Could not create FileID for predefines?");
676 setPredefinesFileID(FID);
677
678 // Start parsing the predefines.
679 EnterSourceFile(FID, nullptr, SourceLocation());
680
681 if (!PPOpts.PCHThroughHeader.empty()) {
682 // Lookup and save the FileID for the through header. If it isn't found
683 // in the search path, it's a fatal error.
685 SourceLocation(), PPOpts.PCHThroughHeader,
686 /*isAngled=*/false, /*FromDir=*/nullptr, /*FromFile=*/nullptr,
687 /*CurDir=*/nullptr, /*SearchPath=*/nullptr, /*RelativePath=*/nullptr,
688 /*SuggestedModule=*/nullptr, /*IsMapped=*/nullptr,
689 /*IsFrameworkFound=*/nullptr);
690 if (!File) {
691 Diag(SourceLocation(), diag::err_pp_through_header_not_found)
692 << PPOpts.PCHThroughHeader;
693 return;
694 }
695 setPCHThroughHeaderFileID(
696 SourceMgr.createFileID(*File, SourceLocation(), SrcMgr::C_User));
697 }
698
699 // Skip tokens from the Predefines and if needed the main file.
700 if ((usingPCHWithThroughHeader() && SkippingUntilPCHThroughHeader) ||
701 (usingPCHWithPragmaHdrStop() && SkippingUntilPragmaHdrStop))
703}
704
705void Preprocessor::EnterDeferredGMFInputs(SourceLocation IncludeLoc) {
706 if (!hasDeferredGMFInputs())
707 return;
708 // Synthesize the implicit input directives and enter them inside the global
709 // module fragment. Attribute the buffer to IncludeLoc so it is ordered within
710 // the translation unit.
711 std::unique_ptr<llvm::MemoryBuffer> MB = llvm::MemoryBuffer::getMemBufferCopy(
712 DeferredGMFInputs, "<gmf-command-line-inputs>");
713 DeferredGMFInputs.clear();
714 DeferredGMFInputsFileID =
715 SourceMgr.createFileID(std::move(MB), SrcMgr::C_User, 0, 0, IncludeLoc);
716 EnterSourceFile(DeferredGMFInputsFileID, nullptr, IncludeLoc);
717}
718
719void Preprocessor::setPCHThroughHeaderFileID(FileID FID) {
720 assert(PCHThroughHeaderFileID.isInvalid() &&
721 "PCHThroughHeaderFileID already set!");
722 PCHThroughHeaderFileID = FID;
723}
724
726 assert(PCHThroughHeaderFileID.isValid() &&
727 "Invalid PCH through header FileID");
728 return FE == SourceMgr.getFileEntryForID(PCHThroughHeaderFileID);
729}
730
732 return TUKind == TU_Prefix && !PPOpts.PCHThroughHeader.empty() &&
733 PCHThroughHeaderFileID.isValid();
734}
735
737 return TUKind != TU_Prefix && !PPOpts.PCHThroughHeader.empty() &&
738 PCHThroughHeaderFileID.isValid();
739}
740
742 return TUKind == TU_Prefix && PPOpts.PCHWithHdrStop;
743}
744
746 return TUKind != TU_Prefix && PPOpts.PCHWithHdrStop;
747}
748
749/// Skip tokens until after the #include of the through header or
750/// until after a #pragma hdrstop is seen. Tokens in the predefines file
751/// and the main file may be skipped. If the end of the predefines file
752/// is reached, skipping continues into the main file. If the end of the
753/// main file is reached, it's a fatal error.
755 bool ReachedMainFileEOF = false;
756 bool UsingPCHThroughHeader = SkippingUntilPCHThroughHeader;
757 bool UsingPragmaHdrStop = SkippingUntilPragmaHdrStop;
758 Token Tok;
759 while (true) {
760 bool InPredefines =
761 (CurLexer && CurLexer->getFileID() == getPredefinesFileID());
762 CurLexerCallback(*this, Tok);
763 if (Tok.is(tok::eof) && !InPredefines) {
764 ReachedMainFileEOF = true;
765 break;
766 }
767 if (UsingPCHThroughHeader && !SkippingUntilPCHThroughHeader)
768 break;
769 if (UsingPragmaHdrStop && !SkippingUntilPragmaHdrStop)
770 break;
771 }
772 if (ReachedMainFileEOF) {
773 if (UsingPCHThroughHeader)
774 Diag(SourceLocation(), diag::err_pp_through_header_not_seen)
775 << PPOpts.PCHThroughHeader << 1;
776 else if (!PPOpts.PCHWithHdrStopCreate)
777 Diag(SourceLocation(), diag::err_pp_pragma_hdrstop_not_seen);
778 }
779}
780
781void Preprocessor::replayPreambleConditionalStack() {
782 // Restore the conditional stack from the preamble, if there is one.
783 if (PreambleConditionalStack.isReplaying()) {
784 assert(CurPPLexer &&
785 "CurPPLexer is null when calling replayPreambleConditionalStack.");
786 CurPPLexer->setConditionalLevels(PreambleConditionalStack.getStack());
787 PreambleConditionalStack.doneReplaying();
788 if (PreambleConditionalStack.reachedEOFWhileSkipping())
789 SkipExcludedConditionalBlock(
790 PreambleConditionalStack.SkipInfo->HashTokenLoc,
791 PreambleConditionalStack.SkipInfo->IfTokenLoc,
792 PreambleConditionalStack.SkipInfo->FoundNonSkipPortion,
793 PreambleConditionalStack.SkipInfo->FoundElse,
794 PreambleConditionalStack.SkipInfo->ElseLoc);
795 }
796}
797
799 // Notify the client that we reached the end of the source file.
800 if (Callbacks)
801 Callbacks->EndOfMainFile();
802}
803
804//===----------------------------------------------------------------------===//
805// Lexer Event Handling.
806//===----------------------------------------------------------------------===//
807
808/// LookUpIdentifierInfo - Given a tok::raw_identifier token, look up the
809/// identifier information for the token and install it into the token,
810/// updating the token kind accordingly.
812 assert(!Identifier.getRawIdentifier().empty() && "No raw identifier data!");
813
814 // Look up this token, see if it is a macro, or if it is a language keyword.
815 IdentifierInfo *II;
816 if (!Identifier.needsCleaning() && !Identifier.hasUCN()) {
817 // No cleaning needed, just use the characters from the lexed buffer.
818 II = getIdentifierInfo(Identifier.getRawIdentifier());
819 } else {
820 // Cleaning needed, alloca a buffer, clean into it, then use the buffer.
821 SmallString<64> IdentifierBuffer;
822 StringRef CleanedStr = getSpelling(Identifier, IdentifierBuffer);
823
824 if (Identifier.hasUCN()) {
825 SmallString<64> UCNIdentifierBuffer;
826 expandUCNs(UCNIdentifierBuffer, CleanedStr);
827 II = getIdentifierInfo(UCNIdentifierBuffer);
828 } else {
829 II = getIdentifierInfo(CleanedStr);
830 }
831 }
832
833 // Update the token info (identifier info and appropriate token kind).
834 // FIXME: the raw_identifier may contain leading whitespace which is removed
835 // from the cleaned identifier token. The SourceLocation should be updated to
836 // refer to the non-whitespace character. For instance, the text "\\\nB" (a
837 // line continuation before 'B') is parsed as a single tok::raw_identifier and
838 // is cleaned to tok::identifier "B". After cleaning the token's length is
839 // still 3 and the SourceLocation refers to the location of the backslash.
840 Identifier.setIdentifierInfo(II);
841 Identifier.setKind(II->getTokenID());
842
843 return II;
844}
845
847 PoisonReasons[II] = DiagID;
848}
849
851 assert(Ident__exception_code && Ident__exception_info);
852 assert(Ident___exception_code && Ident___exception_info);
853 Ident__exception_code->setIsPoisoned(Poison);
854 Ident___exception_code->setIsPoisoned(Poison);
855 Ident_GetExceptionCode->setIsPoisoned(Poison);
856 Ident__exception_info->setIsPoisoned(Poison);
857 Ident___exception_info->setIsPoisoned(Poison);
858 Ident_GetExceptionInfo->setIsPoisoned(Poison);
859 Ident__abnormal_termination->setIsPoisoned(Poison);
860 Ident___abnormal_termination->setIsPoisoned(Poison);
861 Ident_AbnormalTermination->setIsPoisoned(Poison);
862}
863
865 assert(Identifier.getIdentifierInfo() &&
866 "Can't handle identifiers without identifier info!");
867 llvm::DenseMap<IdentifierInfo*,unsigned>::const_iterator it =
868 PoisonReasons.find(Identifier.getIdentifierInfo());
869 if(it == PoisonReasons.end())
870 Diag(Identifier, diag::err_pp_used_poisoned_id);
871 else
872 Diag(Identifier,it->second) << Identifier.getIdentifierInfo();
873}
874
875void Preprocessor::updateOutOfDateIdentifier(const IdentifierInfo &II) const {
876 assert(II.isOutOfDate() && "not out of date");
877 assert(getExternalSource() &&
878 "getExternalSource() should not return nullptr");
880}
881
882/// HandleIdentifier - This callback is invoked when the lexer reads an
883/// identifier. This callback looks up the identifier in the map and/or
884/// potentially macro expands it or turns it into a named token (like 'for').
885///
886/// Note that callers of this method are guarded by checking the
887/// IdentifierInfo's 'isHandleIdentifierCase' bit. If this method changes, the
888/// IdentifierInfo methods that compute these properties will need to change to
889/// match.
891 assert(Identifier.getIdentifierInfo() &&
892 "Can't handle identifiers without identifier info!");
893
894 IdentifierInfo &II = *Identifier.getIdentifierInfo();
895
896 // If the information about this identifier is out of date, update it from
897 // the external source.
898 // We have to treat __VA_ARGS__ in a special way, since it gets
899 // serialized with isPoisoned = true, but our preprocessor may have
900 // unpoisoned it if we're defining a C99 macro.
901 if (II.isOutOfDate()) {
902 bool CurrentIsPoisoned = false;
903 const bool IsSpecialVariadicMacro =
904 &II == Ident__VA_ARGS__ || &II == Ident__VA_OPT__;
905 if (IsSpecialVariadicMacro)
906 CurrentIsPoisoned = II.isPoisoned();
907
908 updateOutOfDateIdentifier(II);
909 Identifier.setKind(II.getTokenID());
910
911 if (IsSpecialVariadicMacro)
912 II.setIsPoisoned(CurrentIsPoisoned);
913 }
914
915 // If this identifier was poisoned, and if it was not produced from a macro
916 // expansion, emit an error.
917 if (II.isPoisoned() && CurPPLexer) {
918 HandlePoisonedIdentifier(Identifier);
919 }
920
921 // If this is a macro to be expanded, do it.
922 if (const MacroDefinition MD = getMacroDefinition(&II)) {
923 const auto *MI = MD.getMacroInfo();
924 assert(MI && "macro definition with no macro info?");
925 if (!DisableMacroExpansion) {
926 if (!Identifier.isExpandDisabled() && MI->isEnabled()) {
927 // C99 6.10.3p10: If the preprocessing token immediately after the
928 // macro name isn't a '(', this macro should not be expanded.
929 if (!MI->isFunctionLike() || isNextPPTokenOneOf(tok::l_paren))
930 return HandleMacroExpandedIdentifier(Identifier, MD);
931 } else {
932 // C99 6.10.3.4p2 says that a disabled macro may never again be
933 // expanded, even if it's in a context where it could be expanded in the
934 // future.
935 Identifier.setFlag(Token::DisableExpand);
936 if (MI->isObjectLike() || isNextPPTokenOneOf(tok::l_paren))
937 Diag(Identifier, diag::pp_disabled_macro_expansion);
938 }
939 }
940 }
941
942 // If this identifier is a keyword in a newer Standard or proposed Standard,
943 // produce a warning. Don't warn if we're not considering macro expansion,
944 // since this identifier might be the name of a macro.
945 // FIXME: This warning is disabled in cases where it shouldn't be, like
946 // "#define constexpr constexpr", "int constexpr;"
947 if (II.isFutureCompatKeyword() && !DisableMacroExpansion) {
948 Diag(Identifier, getIdentifierTable().getFutureCompatDiagKind(II, getLangOpts()))
949 << II.getName();
950 // Don't diagnose this keyword again in this translation unit.
951 II.setIsFutureCompatKeyword(false);
952 }
953
954 // If this identifier would be a keyword in C++, diagnose as a compatibility
955 // issue.
956 if (II.IsKeywordInCPlusPlus() && !DisableMacroExpansion)
957 Diag(Identifier, diag::warn_pp_identifier_is_cpp_keyword) << &II;
958
959 // If this is an extension token, diagnose its use.
960 // We avoid diagnosing tokens that originate from macro definitions.
961 // FIXME: This warning is disabled in cases where it shouldn't be,
962 // like "#define TY typeof", "TY(1) x".
963 if (II.isExtensionToken() && !DisableMacroExpansion)
964 Diag(Identifier, diag::ext_token_used);
965
966 // Handle module contextual keywords.
967 if (getLangOpts().CPlusPlusModules && CurLexer &&
968 !CurLexer->isLexingRawMode() && !CurLexer->isPragmaLexer() &&
969 !CurLexer->ParsingPreprocessorDirective &&
970 Identifier.isModuleContextualKeyword() &&
971 HandleModuleContextualKeyword(Identifier)) {
972 HandleDirective(Identifier);
973 // With a fatal failure in the module loader, we abort parsing.
975 }
976
977 return true;
978}
979
981 ++LexLevel;
982
983 // We loop here until a lex function returns a token; this avoids recursion.
984 while (!CurLexerCallback(*this, Result))
985 ;
986
987 if (Result.is(tok::unknown) && TheModuleLoader.HadFatalFailure)
988 return;
989
990 if (Result.is(tok::code_completion) && Result.getIdentifierInfo()) {
991 // Remember the identifier before code completion token.
992 setCodeCompletionIdentifierInfo(Result.getIdentifierInfo());
993 setCodeCompletionTokenRange(Result.getLocation(), Result.getEndLoc());
994 // Set IdenfitierInfo to null to avoid confusing code that handles both
995 // identifiers and completion tokens.
996 Result.setIdentifierInfo(nullptr);
997 }
998
999 // Update StdCXXImportSeqState to track our position within a C++20 import-seq
1000 // if this token is being produced as a result of phase 4 of translation.
1001 // Update TrackGMFState to decide if we are currently in a Global Module
1002 // Fragment. GMF state updates should precede StdCXXImportSeq ones, since GMF state
1003 // depends on the prevailing StdCXXImportSeq state in two cases.
1004 if (getLangOpts().CPlusPlusModules && LexLevel == 1 &&
1005 !Result.getFlag(Token::IsReinjected)) {
1006 switch (Result.getKind()) {
1007 case tok::l_paren: case tok::l_square: case tok::l_brace:
1008 StdCXXImportSeqState.handleOpenBracket();
1009 break;
1010 case tok::r_paren: case tok::r_square:
1011 StdCXXImportSeqState.handleCloseBracket();
1012 break;
1013 case tok::r_brace:
1014 StdCXXImportSeqState.handleCloseBrace();
1015 break;
1016#define PRAGMA_ANNOTATION(X) case tok::annot_##X:
1017// For `#pragma ...` mimic ';'.
1018#include "clang/Basic/TokenKinds.def"
1019#undef PRAGMA_ANNOTATION
1020 // This token is injected to represent the translation of '#include "a.h"'
1021 // into "import a.h;". Mimic the notional ';'.
1022 case tok::annot_module_include:
1023 case tok::annot_repl_input_end:
1024 case tok::semi:
1025 TrackGMFState.handleSemi();
1026 StdCXXImportSeqState.handleSemi();
1027 ModuleDeclState.handleSemi();
1028 break;
1029 case tok::header_name:
1030 case tok::annot_header_unit:
1031 StdCXXImportSeqState.handleHeaderName();
1032 break;
1033 case tok::kw_export:
1036 TrackGMFState.handleExport();
1037 StdCXXImportSeqState.handleExport();
1038 ModuleDeclState.handleExport();
1039 break;
1040 case tok::colon:
1041 ModuleDeclState.handleColon();
1042 break;
1043 case tok::kw_import:
1044 if (StdCXXImportSeqState.atTopLevel()) {
1045 TrackGMFState.handleImport(StdCXXImportSeqState.afterTopLevelSeq());
1046 StdCXXImportSeqState.handleImport();
1047 }
1048 break;
1049 case tok::kw_module:
1050 if (StdCXXImportSeqState.atTopLevel()) {
1053 TrackGMFState.handleModule(StdCXXImportSeqState.afterTopLevelSeq());
1054 ModuleDeclState.handleModule();
1055 }
1056 break;
1057 case tok::annot_module_name:
1058 ModuleDeclState.handleModuleName(
1059 static_cast<ModuleNameLoc *>(Result.getAnnotationValue()));
1060 if (ModuleDeclState.isModuleCandidate())
1061 break;
1062 [[fallthrough]];
1063 default:
1064 TrackGMFState.handleMisc();
1065 StdCXXImportSeqState.handleMisc();
1066 ModuleDeclState.handleMisc();
1067 break;
1068 }
1069 }
1070
1071 if (RecordCheckPoints && CurLexer &&
1072 ++CheckPointCounter == CheckPointStepSize) {
1073 CheckPoints[CurLexer->getFileID()].push_back(CurLexer->BufferPtr);
1074 CheckPointCounter = 0;
1075 }
1076
1077 if (Result.isNot(tok::kw_export))
1078 LastExportKeyword.startToken();
1079
1080 --LexLevel;
1081
1082 // Destroy any lexers that were deferred while we were in nested Lex calls.
1083 // This must happen after decrementing LexLevel but before any other
1084 // processing that might re-enter Lex.
1085 if (LexLevel == 0 && !PendingDestroyLexers.empty())
1086 PendingDestroyLexers.clear();
1087
1088 if ((LexLevel == 0 || PreprocessToken) &&
1089 !Result.getFlag(Token::IsReinjected)) {
1090 if (LexLevel == 0)
1091 ++TokenCount;
1092 if (OnToken)
1093 OnToken(Result);
1094 }
1095}
1096
1097void Preprocessor::LexTokensUntilEOF(std::vector<Token> *Tokens) {
1098 while (1) {
1099 Token Tok;
1100 Lex(Tok);
1101 if (Tok.isOneOf(tok::unknown, tok::eof, tok::eod,
1102 tok::annot_repl_input_end))
1103 break;
1104 if (Tokens != nullptr)
1105 Tokens->push_back(Tok);
1106 }
1107}
1108
1109/// Lex a header-name token (including one formed from header-name-tokens if
1110/// \p AllowMacroExpansion is \c true).
1111///
1112/// \param FilenameTok Filled in with the next token. On success, this will
1113/// be either a header_name token. On failure, it will be whatever other
1114/// token was found instead.
1115/// \param AllowMacroExpansion If \c true, allow the header name to be formed
1116/// by macro expansion (concatenating tokens as necessary if the first
1117/// token is a '<').
1118/// \return \c true if we reached EOD or EOF while looking for a > token in
1119/// a concatenated header name and diagnosed it. \c false otherwise.
1120bool Preprocessor::LexHeaderName(Token &FilenameTok, bool AllowMacroExpansion) {
1121 // Lex using header-name tokenization rules if tokens are being lexed from
1122 // a file. Just grab a token normally if we're in a macro expansion.
1123 if (CurPPLexer) {
1124 // Avoid nested header-name lexing when macro expansion recurses
1125 // __has_include(__has_include))
1126 if (CurPPLexer->ParsingFilename)
1127 LexUnexpandedToken(FilenameTok);
1128 else
1129 CurPPLexer->LexIncludeFilename(FilenameTok);
1130 } else {
1131 Lex(FilenameTok);
1132 }
1133
1134 // This could be a <foo/bar.h> file coming from a macro expansion. In this
1135 // case, glue the tokens together into an angle_string_literal token.
1136 SmallString<128> FilenameBuffer;
1137 if (FilenameTok.is(tok::less) && AllowMacroExpansion) {
1138 bool StartOfLine = FilenameTok.isAtStartOfLine();
1139 bool LeadingSpace = FilenameTok.hasLeadingSpace();
1140 bool LeadingEmptyMacro = FilenameTok.hasLeadingEmptyMacro();
1141
1142 SourceLocation Start = FilenameTok.getLocation();
1143 SourceLocation End;
1144 FilenameBuffer.push_back('<');
1145
1146 // Consume tokens until we find a '>'.
1147 // FIXME: A header-name could be formed starting or ending with an
1148 // alternative token. It's not clear whether that's ill-formed in all
1149 // cases.
1150 while (FilenameTok.isNot(tok::greater)) {
1151 Lex(FilenameTok);
1152 if (FilenameTok.isOneOf(tok::eod, tok::eof)) {
1153 Diag(FilenameTok.getLocation(), diag::err_expected) << tok::greater;
1154 Diag(Start, diag::note_matching) << tok::less;
1155 return true;
1156 }
1157
1158 End = FilenameTok.getLocation();
1159
1160 // FIXME: Provide code completion for #includes.
1161 if (FilenameTok.is(tok::code_completion)) {
1163 Lex(FilenameTok);
1164 continue;
1165 }
1166
1167 // Append the spelling of this token to the buffer. If there was a space
1168 // before it, add it now.
1169 if (FilenameTok.hasLeadingSpace())
1170 FilenameBuffer.push_back(' ');
1171
1172 // Get the spelling of the token, directly into FilenameBuffer if
1173 // possible.
1174 size_t PreAppendSize = FilenameBuffer.size();
1175 FilenameBuffer.resize(PreAppendSize + FilenameTok.getLength());
1176
1177 const char *BufPtr = &FilenameBuffer[PreAppendSize];
1178 unsigned ActualLen = getSpelling(FilenameTok, BufPtr);
1179
1180 // If the token was spelled somewhere else, copy it into FilenameBuffer.
1181 if (BufPtr != &FilenameBuffer[PreAppendSize])
1182 memcpy(&FilenameBuffer[PreAppendSize], BufPtr, ActualLen);
1183
1184 // Resize FilenameBuffer to the correct size.
1185 if (FilenameTok.getLength() != ActualLen)
1186 FilenameBuffer.resize(PreAppendSize + ActualLen);
1187 }
1188
1189 FilenameTok.startToken();
1190 FilenameTok.setKind(tok::header_name);
1191 FilenameTok.setFlagValue(Token::StartOfLine, StartOfLine);
1192 FilenameTok.setFlagValue(Token::LeadingSpace, LeadingSpace);
1193 FilenameTok.setFlagValue(Token::LeadingEmptyMacro, LeadingEmptyMacro);
1194 CreateString(FilenameBuffer, FilenameTok, Start, End);
1195 } else if (FilenameTok.is(tok::string_literal) && AllowMacroExpansion) {
1196 // Convert a string-literal token of the form " h-char-sequence "
1197 // (produced by macro expansion) into a header-name token.
1198 //
1199 // The rules for header-names don't quite match the rules for
1200 // string-literals, but all the places where they differ result in
1201 // undefined behavior, so we can and do treat them the same.
1202 //
1203 // A string-literal with a prefix or suffix is not translated into a
1204 // header-name. This could theoretically be observable via the C++20
1205 // context-sensitive header-name formation rules.
1206 StringRef Str = getSpelling(FilenameTok, FilenameBuffer);
1207 if (Str.size() >= 2 && Str.front() == '"' && Str.back() == '"')
1208 FilenameTok.setKind(tok::header_name);
1209 }
1210
1211 return false;
1212}
1213
1214std::optional<Token> Preprocessor::peekNextPPToken() const {
1215 // Do some quick tests for rejection cases.
1216 std::optional<Token> Val;
1217 if (CurLexer)
1218 Val = CurLexer->peekNextPPToken();
1219 else
1220 Val = CurTokenLexer->peekNextPPToken();
1221
1222 if (!Val) {
1223 // We have run off the end. If it's a source file we don't
1224 // examine enclosing ones (C99 5.1.1.2p4). Otherwise walk up the
1225 // macro stack.
1226 if (CurPPLexer)
1227 return std::nullopt;
1228 for (const IncludeStackInfo &Entry : llvm::reverse(IncludeMacroStack)) {
1229 if (Entry.TheLexer)
1230 Val = Entry.TheLexer->peekNextPPToken();
1231 else
1232 Val = Entry.TheTokenLexer->peekNextPPToken();
1233
1234 if (Val)
1235 break;
1236
1237 // Ran off the end of a source file?
1238 if (Entry.ThePPLexer)
1239 return std::nullopt;
1240 }
1241 }
1242
1243 // Okay, we found the token and return. Otherwise we found the end of the
1244 // translation unit.
1245 return Val;
1246}
1247
1248// We represent the primary and partition names as 'Paths' which are sections
1249// of the hierarchical access path for a clang module. However for C++20
1250// the periods in a name are just another character, and we will need to
1251// flatten them into a string.
1253 std::string Name;
1254 if (Path.empty())
1255 return Name;
1256
1257 for (auto &Piece : Path) {
1258 assert(Piece.getIdentifierInfo() && Piece.getLoc().isValid());
1259 if (!Name.empty())
1260 Name += ".";
1261 Name += Piece.getIdentifierInfo()->getName();
1262 }
1263 return Name;
1264}
1265
1267 assert(!Path.empty() && "expect at least one identifier in a module name");
1268 void *Mem = PP.getPreprocessorAllocator().Allocate(
1269 totalSizeToAlloc<IdentifierLoc>(Path.size()), alignof(ModuleNameLoc));
1270 return new (Mem) ModuleNameLoc(Path);
1271}
1272
1274 SmallVectorImpl<Token> &Suffix,
1276 bool AllowMacroExpansion,
1277 bool IsPartition) {
1278 auto ConsumeToken = [&]() {
1279 if (AllowMacroExpansion)
1280 Lex(Tok);
1281 else
1283 Suffix.push_back(Tok);
1284 };
1285
1286 while (true) {
1287 if (Tok.isNot(tok::identifier)) {
1288 if (Tok.is(tok::code_completion)) {
1289 CurLexer->cutOffLexing();
1290 CodeComplete->CodeCompleteModuleImport(UseLoc, Path);
1291 return true;
1292 }
1293
1294 Diag(Tok, diag::err_pp_module_expected_ident) << Path.empty();
1295 return true;
1296 }
1297
1298 // [cpp.pre]/p2:
1299 // No identifier in the pp-module-name or pp-module-partition shall
1300 // currently be defined as an object-like macro.
1301 if (MacroInfo *MI = getMacroInfo(Tok.getIdentifierInfo());
1302 MI && MI->isObjectLike() && getLangOpts().CPlusPlus20 &&
1303 !AllowMacroExpansion) {
1304 Diag(Tok, diag::err_pp_module_name_is_macro)
1305 << IsPartition << Tok.getIdentifierInfo();
1306 Diag(MI->getDefinitionLoc(), diag::note_macro_here)
1307 << Tok.getIdentifierInfo();
1308 }
1309
1310 // Record this part of the module path.
1311 Path.emplace_back(Tok.getLocation(), Tok.getIdentifierInfo());
1312 ConsumeToken();
1313
1314 if (Tok.isNot(tok::period))
1315 return false;
1316
1317 ConsumeToken();
1318 }
1319}
1320
1321bool Preprocessor::HandleModuleName(StringRef DirType, SourceLocation UseLoc,
1322 Token &Tok,
1324 SmallVectorImpl<Token> &DirToks,
1325 bool AllowMacroExpansion,
1326 bool IsPartition) {
1327 bool LeadingSpace = Tok.hasLeadingSpace();
1328 unsigned NumToksInDirective = DirToks.size();
1329 if (LexModuleNameContinue(Tok, UseLoc, DirToks, Path, AllowMacroExpansion,
1330 IsPartition)) {
1331 if (Tok.isNot(tok::eod))
1332 CheckEndOfDirective(DirType,
1333 /*EnableMacros=*/false, &DirToks);
1335 return true;
1336 }
1337
1338 // Clean the module-name tokens and replace these tokens with
1339 // annot_module_name.
1340 DirToks.resize(NumToksInDirective);
1341 ModuleNameLoc *NameLoc = ModuleNameLoc::Create(*this, Path);
1342 DirToks.emplace_back();
1343 DirToks.back().setKind(tok::annot_module_name);
1344 DirToks.back().setAnnotationRange(NameLoc->getRange());
1345 DirToks.back().setAnnotationValue(static_cast<void *>(NameLoc));
1346 DirToks.back().setFlagValue(Token::LeadingSpace, LeadingSpace);
1347 DirToks.push_back(Tok);
1348 return false;
1349}
1350
1351/// [cpp.pre]/p2:
1352/// A preprocessing directive consists of a sequence of preprocessing tokens
1353/// that satisfies the following constraints: At the start of translation phase
1354/// 4, the first preprocessing token in the sequence, referred to as a
1355/// directive-introducing token, begins with the first character in the source
1356/// file (optionally after whitespace containing no new-line characters) or
1357/// follows whitespace containing at least one new-line character, and is:
1358/// - a # preprocessing token, or
1359/// - an import preprocessing token immediately followed on the same logical
1360/// source line by a header-name, <, identifier, or : preprocessing token, or
1361/// - a module preprocessing token immediately followed on the same logical
1362/// source line by an identifier, :, or ; preprocessing token, or
1363/// - an export preprocessing token immediately followed on the same logical
1364/// source line by one of the two preceding forms.
1365///
1366///
1367/// At the start of phase 4 an import or module token is treated as starting a
1368/// directive and are converted to their respective keywords iff:
1369/// - After skipping horizontal whitespace are
1370/// - at the start of a logical line, or
1371/// - preceded by an 'export' at the start of the logical line.
1372/// - Are followed by an identifier pp token (before macro expansion), or
1373/// - <, ", or : (but not ::) pp tokens for 'import', or
1374/// - ; for 'module'
1375/// Otherwise the token is treated as an identifier.
1377 if (!getLangOpts().CPlusPlusModules || !Result.isModuleContextualKeyword())
1378 return false;
1379
1380 if (Result.is(tok::kw_export)) {
1381 LastExportKeyword = Result;
1382 return false;
1383 }
1384
1385 /// Trait 'module' and 'import' as a identifier when the main file is a
1386 /// preprocessed module file. We only allow '__preprocessed_module' and
1387 /// '__preprocessed_import' in this context.
1388 IdentifierInfo *II = Result.getIdentifierInfo();
1390 (II->isStr(tok::getKeywordSpelling(tok::kw_import)) ||
1391 II->isStr(tok::getKeywordSpelling(tok::kw_module))))
1392 return false;
1393
1394 if (LastExportKeyword.is(tok::kw_export)) {
1395 // The export keyword was not at the start of line, it's not a
1396 // directive-introducing token.
1397 if (!LastExportKeyword.isAtPhysicalStartOfLine())
1398 return false;
1399 // [cpp.pre]/1.4
1400 // export // not a preprocessing directive
1401 // import foo; // preprocessing directive (ill-formed at phase7)
1402 if (Result.isAtPhysicalStartOfLine())
1403 return false;
1404 } else if (!Result.isAtPhysicalStartOfLine())
1405 return false;
1406
1407 assert(CurPPLexer && "CurPPLexer must not be null");
1408
1409 llvm::SaveAndRestore<bool> SavedParsingPreprocessorDirective(
1410 CurPPLexer->ParsingPreprocessorDirective, true);
1411
1412 if (II->isModuleKeyword()) {
1413 if (auto NextTok = peekNextPPToken()) {
1414 if (NextTok->is(tok::raw_identifier))
1415 LookUpIdentifierInfo(*NextTok);
1416 if (NextTok->isOneOf(tok::identifier, tok::colon, tok::semi)) {
1417 Result.setKind(tok::kw_module);
1418 ModuleDeclLoc = Result.getLocation();
1419 return true;
1420 }
1421 }
1422 return false;
1423 }
1424
1425 if (II->isImportKeyword()) {
1426 llvm::SaveAndRestore<bool> SavedParsingFilename(CurPPLexer->ParsingFilename,
1427 true);
1428 if (auto NextTok = peekNextPPToken()) {
1429 if (NextTok->is(tok::raw_identifier))
1430 LookUpIdentifierInfo(*NextTok);
1431 if (NextTok->isOneOf(tok::header_name, tok::identifier, tok::colon,
1432 tok::less, tok::code_completion)) {
1433 Result.setKind(tok::kw_import);
1434 ModuleImportLoc = Result.getLocation();
1435 return true;
1436 }
1437 }
1438 return false;
1439 }
1440
1441 // Ok, it's an identifier.
1442 return false;
1443}
1444
1446 SmallVectorImpl<Token> &Toks, bool StopUntilEOD) {
1449 return false;
1450}
1451
1452/// Collect the tokens of a C++20 pp-import-suffix.
1454 bool StopUntilEOD) {
1455 while (true) {
1456 Toks.emplace_back();
1457 Lex(Toks.back());
1458
1459 switch (Toks.back().getKind()) {
1460 case tok::semi:
1461 if (!StopUntilEOD)
1462 return;
1463 [[fallthrough]];
1464 case tok::eod:
1465 case tok::eof:
1466 return;
1467 default:
1468 break;
1469 }
1470 }
1471}
1472
1473// Allocate a holding buffer for a sequence of tokens and introduce it into
1474// the token stream.
1476 if (Toks.empty())
1477 return;
1478 auto ToksCopy = std::make_unique<Token[]>(Toks.size());
1479 std::copy(Toks.begin(), Toks.end(), ToksCopy.get());
1480 EnterTokenStream(std::move(ToksCopy), Toks.size(),
1481 /*DisableMacroExpansion*/ false, /*IsReinject*/ false);
1482 assert(CurTokenLexer && "Must have a TokenLexer");
1483 CurTokenLexer->setLexingCXXModuleDirective();
1484}
1485
1487 bool IncludeExports) {
1488 CurSubmoduleState->VisibleModules.setVisible(
1489 M, Loc, IncludeExports, [](Module *) {},
1490 [&](ArrayRef<Module *> Path, Module *Conflict, StringRef Message) {
1491 // FIXME: Include the path in the diagnostic.
1492 // FIXME: Include the import location for the conflicting module.
1493 Diag(ModuleImportLoc, diag::warn_module_conflict)
1494 << Path[0]->getFullModuleName()
1495 << Conflict->getFullModuleName()
1496 << Message;
1497 });
1498
1499 // Add this module to the imports list of the currently-built submodule.
1500 if (!BuildingSubmoduleStack.empty() && M != BuildingSubmoduleStack.back().M)
1501 BuildingSubmoduleStack.back().M->Imports.push_back(M);
1502}
1503
1505 const char *DiagnosticTag,
1506 bool AllowMacroExpansion) {
1507 // We need at least one string literal.
1508 if (Result.isNot(tok::string_literal)) {
1509 Diag(Result, diag::err_expected_string_literal)
1510 << /*Source='in...'*/0 << DiagnosticTag;
1511 return false;
1512 }
1513
1514 // Lex string literal tokens, optionally with macro expansion.
1515 SmallVector<Token, 4> StrToks;
1516 do {
1517 StrToks.push_back(Result);
1518
1519 if (Result.hasUDSuffix())
1520 Diag(Result, diag::err_invalid_string_udl);
1521
1522 if (AllowMacroExpansion)
1523 Lex(Result);
1524 else
1526 } while (Result.is(tok::string_literal));
1527
1528 // Concatenate and parse the strings.
1529 StringLiteralParser Literal(StrToks, *this);
1530 assert(Literal.isOrdinary() && "Didn't allow wide strings in");
1531
1532 if (Literal.hadError)
1533 return false;
1534
1535 if (Literal.Pascal) {
1536 Diag(StrToks[0].getLocation(), diag::err_expected_string_literal)
1537 << /*Source='in...'*/0 << DiagnosticTag;
1538 return false;
1539 }
1540
1541 String = std::string(Literal.GetString());
1542 return true;
1543}
1544
1546 assert(Tok.is(tok::numeric_constant));
1547 SmallString<8> IntegerBuffer;
1548 bool NumberInvalid = false;
1549 StringRef Spelling = getSpelling(Tok, IntegerBuffer, &NumberInvalid);
1550 if (NumberInvalid)
1551 return false;
1552 NumericLiteralParser Literal(Spelling, Tok.getLocation(), getSourceManager(),
1554 getDiagnostics());
1555 if (Literal.hadError || !Literal.isIntegerLiteral() || Literal.hasUDSuffix())
1556 return false;
1557 llvm::APInt APVal(64, 0);
1558 if (Literal.GetIntegerValue(APVal))
1559 return false;
1560 Lex(Tok);
1561 Value = APVal.getLimitedValue();
1562 return true;
1563}
1564
1566 assert(Handler && "NULL comment handler");
1567 assert(!llvm::is_contained(CommentHandlers, Handler) &&
1568 "Comment handler already registered");
1569 CommentHandlers.push_back(Handler);
1570}
1571
1573 std::vector<CommentHandler *>::iterator Pos =
1574 llvm::find(CommentHandlers, Handler);
1575 assert(Pos != CommentHandlers.end() && "Comment handler not registered");
1576 CommentHandlers.erase(Pos);
1577}
1578
1580 bool AnyPendingTokens = false;
1581 for (CommentHandler *H : CommentHandlers) {
1582 if (H->HandleComment(*this, Comment))
1583 AnyPendingTokens = true;
1584 }
1585 if (!AnyPendingTokens || getCommentRetentionState())
1586 return false;
1587 Lex(result);
1588 return true;
1589}
1590
1591void Preprocessor::emitMacroDeprecationWarning(const Token &Identifier) const {
1592 const MacroAnnotations &A =
1594 assert(A.DeprecationInfo &&
1595 "Macro deprecation warning without recorded annotation!");
1596 const MacroAnnotationInfo &Info = *A.DeprecationInfo;
1597 if (Info.Message.empty())
1598 Diag(Identifier, diag::warn_pragma_deprecated_macro_use)
1599 << Identifier.getIdentifierInfo() << 0;
1600 else
1601 Diag(Identifier, diag::warn_pragma_deprecated_macro_use)
1602 << Identifier.getIdentifierInfo() << 1 << Info.Message;
1603 Diag(Info.Location, diag::note_pp_macro_annotation) << 0;
1604}
1605
1606void Preprocessor::emitRestrictExpansionWarning(const Token &Identifier) const {
1607 const MacroAnnotations &A =
1609 assert(A.RestrictExpansionInfo &&
1610 "Macro restricted expansion warning without recorded annotation!");
1611 const MacroAnnotationInfo &Info = *A.RestrictExpansionInfo;
1612 if (Info.Message.empty())
1613 Diag(Identifier, diag::warn_pragma_restrict_expansion_macro_use)
1614 << Identifier.getIdentifierInfo() << 0;
1615 else
1616 Diag(Identifier, diag::warn_pragma_restrict_expansion_macro_use)
1617 << Identifier.getIdentifierInfo() << 1 << Info.Message;
1618 Diag(Info.Location, diag::note_pp_macro_annotation) << 1;
1619}
1620
1621void Preprocessor::emitRestrictInfNaNWarning(const Token &Identifier,
1622 unsigned DiagSelection) const {
1623 Diag(Identifier, diag::warn_fp_nan_inf_when_disabled) << DiagSelection << 1;
1624}
1625
1626void Preprocessor::emitFinalMacroWarning(const Token &Identifier,
1627 bool IsUndef) const {
1628 const MacroAnnotations &A =
1630 assert(A.FinalAnnotationLoc &&
1631 "Final macro warning without recorded annotation!");
1632
1633 Diag(Identifier, diag::warn_pragma_final_macro)
1634 << Identifier.getIdentifierInfo() << (IsUndef ? 0 : 1);
1635 Diag(*A.FinalAnnotationLoc, diag::note_pp_macro_annotation) << 2;
1636}
1637
1639 const SourceLocation &Loc) const {
1640 // The lambda that tests if a `Loc` is in an opt-out region given one opt-out
1641 // region map:
1642 auto TestInMap = [&SourceMgr](const SafeBufferOptOutRegionsTy &Map,
1643 const SourceLocation &Loc) -> bool {
1644 // Try to find a region in `SafeBufferOptOutMap` where `Loc` is in:
1645 auto FirstRegionEndingAfterLoc = llvm::partition_point(
1646 Map, [&SourceMgr,
1647 &Loc](const std::pair<SourceLocation, SourceLocation> &Region) {
1648 return SourceMgr.isBeforeInTranslationUnit(Region.second, Loc);
1649 });
1650
1651 if (FirstRegionEndingAfterLoc != Map.end()) {
1652 // To test if the start location of the found region precedes `Loc`:
1653 return SourceMgr.isBeforeInTranslationUnit(
1654 FirstRegionEndingAfterLoc->first, Loc);
1655 }
1656 // If we do not find a region whose end location passes `Loc`, we want to
1657 // check if the current region is still open:
1658 if (!Map.empty() && Map.back().first == Map.back().second)
1659 return SourceMgr.isBeforeInTranslationUnit(Map.back().first, Loc);
1660 return false;
1661 };
1662
1663 // What the following does:
1664 //
1665 // If `Loc` belongs to the local TU, we just look up `SafeBufferOptOutMap`.
1666 // Otherwise, `Loc` is from a loaded AST. We look up the
1667 // `LoadedSafeBufferOptOutMap` first to get the opt-out region map of the
1668 // loaded AST where `Loc` is at. Then we find if `Loc` is in an opt-out
1669 // region w.r.t. the region map. If the region map is absent, it means there
1670 // is no opt-out pragma in that loaded AST.
1671 //
1672 // Opt-out pragmas in the local TU or a loaded AST is not visible to another
1673 // one of them. That means if you put the pragmas around a `#include
1674 // "module.h"`, where module.h is a module, it is not actually suppressing
1675 // warnings in module.h. This is fine because warnings in module.h will be
1676 // reported when module.h is compiled in isolation and nothing in module.h
1677 // will be analyzed ever again. So you will not see warnings from the file
1678 // that imports module.h anyway. And you can't even do the same thing for PCHs
1679 // because they can only be included from the command line.
1680
1681 if (SourceMgr.isLocalSourceLocation(Loc))
1682 return TestInMap(SafeBufferOptOutMap, Loc);
1683
1684 const SafeBufferOptOutRegionsTy *LoadedRegions =
1685 LoadedSafeBufferOptOutMap.lookupLoadedOptOutMap(Loc, SourceMgr);
1686
1687 if (LoadedRegions)
1688 return TestInMap(*LoadedRegions, Loc);
1689 return false;
1690}
1691
1693 bool isEnter, const SourceLocation &Loc) {
1694 if (isEnter) {
1696 return true; // invalid enter action
1697 InSafeBufferOptOutRegion = true;
1698 CurrentSafeBufferOptOutStart = Loc;
1699
1700 // To set the start location of a new region:
1701
1702 if (!SafeBufferOptOutMap.empty()) {
1703 [[maybe_unused]] auto *PrevRegion = &SafeBufferOptOutMap.back();
1704 assert(PrevRegion->first != PrevRegion->second &&
1705 "Shall not begin a safe buffer opt-out region before closing the "
1706 "previous one.");
1707 }
1708 // If the start location equals to the end location, we call the region a
1709 // open region or a unclosed region (i.e., end location has not been set
1710 // yet).
1711 SafeBufferOptOutMap.emplace_back(Loc, Loc);
1712 } else {
1714 return true; // invalid enter action
1715 InSafeBufferOptOutRegion = false;
1716
1717 // To set the end location of the current open region:
1718
1719 assert(!SafeBufferOptOutMap.empty() &&
1720 "Misordered safe buffer opt-out regions");
1721 auto *CurrRegion = &SafeBufferOptOutMap.back();
1722 assert(CurrRegion->first == CurrRegion->second &&
1723 "Set end location to a closed safe buffer opt-out region");
1724 CurrRegion->second = Loc;
1725 }
1726 return false;
1727}
1728
1730 return InSafeBufferOptOutRegion;
1731}
1733 StartLoc = CurrentSafeBufferOptOutStart;
1734 return InSafeBufferOptOutRegion;
1735}
1736
1739 assert(!InSafeBufferOptOutRegion &&
1740 "Attempt to serialize safe buffer opt-out regions before file being "
1741 "completely preprocessed");
1742
1744
1745 for (const auto &[begin, end] : SafeBufferOptOutMap) {
1746 SrcSeq.push_back(begin);
1747 SrcSeq.push_back(end);
1748 }
1749 // Only `SafeBufferOptOutMap` gets serialized. No need to serialize
1750 // `LoadedSafeBufferOptOutMap` because if this TU loads a pch/module, every
1751 // pch/module in the pch-chain/module-DAG will be loaded one by one in order.
1752 // It means that for each loading pch/module m, it just needs to load m's own
1753 // `SafeBufferOptOutMap`.
1754 return SrcSeq;
1755}
1756
1758 const SmallVectorImpl<SourceLocation> &SourceLocations) {
1759 if (SourceLocations.size() == 0)
1760 return false;
1761
1762 assert(SourceLocations.size() % 2 == 0 &&
1763 "ill-formed SourceLocation sequence");
1764
1765 auto It = SourceLocations.begin();
1766 SafeBufferOptOutRegionsTy &Regions =
1767 LoadedSafeBufferOptOutMap.findAndConsLoadedOptOutMap(*It, SourceMgr);
1768
1769 do {
1770 SourceLocation Begin = *It++;
1771 SourceLocation End = *It++;
1772
1773 Regions.emplace_back(Begin, End);
1774 } while (It != SourceLocations.end());
1775 return true;
1776}
1777
1778ModuleLoader::~ModuleLoader() = default;
1779
1781
1783
1785
1787 if (Record)
1788 return;
1789
1790 Record = new PreprocessingRecord(getSourceManager());
1791 addPPCallbacks(std::unique_ptr<PPCallbacks>(Record));
1792}
1793
1795 auto IsPreserved = [&](PPCallbacks *C) {
1796 return C == Record || C == DirTracer;
1797 };
1799 PPCallbacks::releaseIfPreserved(Callbacks, IsPreserved, Released);
1800 Callbacks.reset();
1801 for (auto *P : Released)
1802 addPPCallbacks(std::unique_ptr<PPCallbacks>(P));
1803}
1804
1805const char *Preprocessor::getCheckPoint(FileID FID, const char *Start) const {
1806 if (auto It = CheckPoints.find(FID); It != CheckPoints.end()) {
1807 const SmallVector<const char *> &FileCheckPoints = It->second;
1808 auto P = llvm::upper_bound(FileCheckPoints, Start);
1809 if (P == FileCheckPoints.begin())
1810 return nullptr;
1811 return *std::prev(P);
1812 }
1813 return nullptr;
1814}
1815
1817 return DirTracer && DirTracer->hasSeenNoTrivialPPDirective();
1818}
1819
1821 return SeenNoTrivialPPDirective;
1822}
1823
1824void NoTrivialPPDirectiveTracer::setSeenNoTrivialPPDirective() {
1825 if (InMainFile && !SeenNoTrivialPPDirective)
1826 SeenNoTrivialPPDirective = true;
1827}
1828
1830 FileID FID, LexedFileChangeReason Reason,
1831 SrcMgr::CharacteristicKind FileType, FileID PrevFID, SourceLocation Loc) {
1832 InMainFile = (FID == PP.getSourceManager().getMainFileID());
1833}
1834
1836 const MacroDefinition &MD,
1837 SourceRange Range,
1838 const MacroArgs *Args) {
1839 // FIXME: Does only enable builtin macro expansion make sense?
1840 if (!MD.getMacroInfo()->isBuiltinMacro())
1841 setSeenNoTrivialPPDirective();
1842}
Defines enum values for all the target-independent builtin functions.
This is the interface for scanning header and source files to get the minimum necessary preprocessor ...
LLVM_INSTANTIATE_REGISTRY_EX(CLANG_ABI_EXPORT, clang::tooling::ToolExecutorPluginRegistry) namespace clang
Definition Execution.cpp:14
Defines the clang::FileManager interface and associated types.
unsigned ColumnWidth
The width of the non-whitespace parts of the token (or its first line for multi-line tokens) in colum...
Token Tok
The Token.
Defines the clang::IdentifierInfo, clang::IdentifierTable, and clang::Selector interfaces.
Forward-declares and imports various common LLVM datatypes that clang wants to use unqualified.
Defines the clang::LangOptions interface.
Defines the clang::MacroInfo and clang::MacroDirective classes.
Defines the clang::Module class, which describes a module in the source code.
Defines the PreprocessorLexer interface.
static bool MacroDefinitionEquals(const MacroInfo *MI, ArrayRef< TokenValue > Tokens)
Compares macro tokens with a specified token value sequence.
static constexpr unsigned CheckPointStepSize
Minimum distance between two check points, in tokens.
Defines the clang::Preprocessor interface.
Defines the clang::SourceLocation class and associated facilities.
Defines the SourceManager interface.
__DEVICE__ void * memcpy(void *__a, const void *__b, size_t __c)
Abstract base class that describes a handler that will receive source ranges for each of the comments...
Concrete class used by the front-end to report problems and issues.
Definition Diagnostic.h:234
virtual void updateOutOfDateIdentifier(const IdentifierInfo &II)=0
Update an out-of-date identifier.
A reference to a FileEntry that includes the name of the file as it was accessed by the FileManager's...
Definition FileEntry.h:57
Cached information about one file (either on disk or in the virtual file system).
Definition FileEntry.h:273
An opaque identifier used by SourceManager which refers to a source file (MemoryBuffer) along with it...
bool isValid() const
bool isInvalid() const
Encapsulates the information needed to find the file referenced by a #include or #include_next,...
Module * lookupModule(StringRef ModuleName, SourceLocation ImportLoc=SourceLocation(), bool AllowSearch=true, bool AllowExtraModuleMapSearch=false)
Lookup a module Search for a module with the given name.
Provides lookups to, and iteration over, IdentiferInfo objects.
One of these records is kept for each identifier that is lexed.
bool IsKeywordInCPlusPlus() const
Return true if this identifier would be a keyword in C++ mode.
bool isModuleKeyword() const
Determine whether this is the contextual keyword module.
tok::TokenKind getTokenID() const
If this is a source-language token (e.g.
void setIsPoisoned(bool Value=true)
setIsPoisoned - Mark this identifier as poisoned.
bool isPoisoned() const
Return true if this token has been poisoned.
bool isImportKeyword() const
Determine whether this is the contextual keyword import.
bool isStr(const char(&Str)[StrLen]) const
Return true if this is the identifier for the specified string.
bool isOutOfDate() const
Determine whether the information for this identifier is out of date with respect to the external sou...
void setIsFutureCompatKeyword(bool Val)
StringRef getName() const
Return the actual identifier string.
bool isFutureCompatKeyword() const
is/setIsFutureCompatKeyword - Initialize information about whether or not this language token is a ke...
bool isExtensionToken() const
get/setExtension - Initialize information about whether or not this language token is an extension.
@ FEM_UnsetOnCommandLine
Used only for FE option processing; this is only used to indicate that the user did not specify an ex...
Keeps track of the various options that can be enabled, which controls the dialect of C or C++ that i...
MacroArgs - An instance of this class captures information about the formal arguments specified to a ...
Definition MacroArgs.h:30
A description of the current definition of a macro.
Definition MacroInfo.h:596
MacroInfo * getMacroInfo() const
Get the MacroInfo that should be used for this definition.
Definition MacroInfo.h:612
SourceLocation getLocation() const
Definition MacroInfo.h:489
Encapsulates the data about a macro definition (e.g.
Definition MacroInfo.h:40
const_tokens_iterator tokens_begin() const
Definition MacroInfo.h:245
unsigned getNumTokens() const
Return the number of tokens that this macro expands to.
Definition MacroInfo.h:236
const Token & getReplacementToken(unsigned Tok) const
Definition MacroInfo.h:238
bool isBuiltinMacro() const
Return true if this macro requires processing before expansion.
Definition MacroInfo.h:218
bool isObjectLike() const
Definition MacroInfo.h:203
Abstract interface for a module loader.
virtual ~ModuleLoader()
static std::string getFlatNameFromPath(ModuleIdPath Path)
Represents a macro directive exported by a module.
Definition MacroInfo.h:515
static ModuleNameLoc * Create(Preprocessor &PP, ModuleIdPath Path)
SourceRange getRange() const
Describes a module or submodule.
Definition Module.h:340
void MacroExpands(const Token &MacroNameTok, const MacroDefinition &MD, SourceRange Range, const MacroArgs *Args) override
Called by Preprocessor::HandleMacroExpandedIdentifier when a macro invocation is found.
void LexedFileChanged(FileID FID, LexedFileChangeReason Reason, SrcMgr::CharacteristicKind FileType, FileID PrevFID, SourceLocation Loc) override
Callback invoked whenever the Lexer moves to a different file for lexing.
NumericLiteralParser - This performs strict semantic analysis of the content of a ppnumber,...
This interface provides a way to observe the actions of the preprocessor as it does its thing.
Definition PPCallbacks.h:37
static void releaseIfPreserved(std::unique_ptr< PPCallbacks > &CB, llvm::function_ref< bool(PPCallbacks *)> Pred, SmallVectorImpl< PPCallbacks * > &Released)
Walk the subtree rooted at CB (recursing into descendants first), then check CB itself.
PragmaNamespace - This PragmaHandler subdivides the namespace of pragmas, allowing hierarchical pragm...
Definition Pragma.h:96
A record of the steps taken while preprocessing a source file, including the various preprocessing di...
void setConditionalLevels(ArrayRef< PPConditionalInfo > CL)
PreprocessorOptions - This class is used for passing the various options used in preprocessor initial...
std::string PCHThroughHeader
If non-empty, the filename used in an include directive in the primary source file (or command-line p...
bool GeneratePreamble
True indicates that a preamble is being generated.
Engages in a tight little dance with the lexer to efficiently preprocess tokens.
bool markIncluded(FileEntryRef File)
Mark the file as included.
void FinalizeForModelFile()
Cleanup after model file parsing.
bool FinishLexStringLiteral(Token &Result, std::string &String, const char *DiagnosticTag, bool AllowMacroExpansion)
Complete the lexing of a string literal where the first token has already been lexed (see LexStringLi...
bool creatingPCHWithThroughHeader()
True if creating a PCH with a through header.
void DumpToken(const Token &Tok, bool DumpFlags=false) const
Print the token to stderr, used for debugging.
void EnterModuleSuffixTokenStream(ArrayRef< Token > Toks)
void InitializeForModelFile()
Initialize the preprocessor to parse a model file.
const MacroInfo * getMacroInfo(const IdentifierInfo *II) const
void setCodeCompletionTokenRange(const SourceLocation Start, const SourceLocation End)
Set the code completion token range for detecting replacement range later on.
void CreateString(StringRef Str, Token &Tok, SourceLocation ExpansionLocStart=SourceLocation(), SourceLocation ExpansionLocEnd=SourceLocation())
Plop the specified string into a scratch buffer and set the specified token's location and length to ...
bool isSafeBufferOptOut(const SourceManager &SourceMgr, const SourceLocation &Loc) const
const char * getCheckPoint(FileID FID, const char *Start) const
Returns a pointer into the given file's buffer that's guaranteed to be between tokens.
IdentifierInfo * LookUpIdentifierInfo(Token &Identifier) const
Given a tok::raw_identifier token, look up the identifier information for the token and install it in...
friend class MacroArgs
void DumpMacro(const MacroInfo &MI) const
llvm::iterator_range< macro_iterator > macros(bool IncludeExternalMacros=true) const
void setCodeCompletionReached()
Note that we hit the code-completion point.
bool SetCodeCompletionPoint(FileEntryRef File, unsigned Line, unsigned Column)
Specify the point at which code-completion will be performed.
void Lex(Token &Result)
Lex the next token for this preprocessor.
const TranslationUnitKind TUKind
The kind of translation unit we are processing.
bool EnterSourceFile(FileID FID, ConstSearchDirIterator Dir, SourceLocation Loc, bool IsFirstIncludeOfFile=true)
Add a source file to the top of the include stack and start lexing tokens from it instead of the curr...
void addCommentHandler(CommentHandler *Handler)
Add the specified comment handler to the preprocessor.
void removeCommentHandler(CommentHandler *Handler)
Remove the specified comment handler.
void HandlePoisonedIdentifier(Token &Identifier)
Display reason for poisoned identifier.
bool HandleIdentifier(Token &Identifier)
Callback invoked when the lexer reads an identifier and has filled in the tokens IdentifierInfo membe...
void addPPCallbacks(std::unique_ptr< PPCallbacks > C)
bool enterOrExitSafeBufferOptOutRegion(bool isEnter, const SourceLocation &Loc)
Alter the state of whether this PP currently is in a "-Wunsafe-buffer-usage" opt-out region.
void EnterMainSourceFile()
Enter the specified FileID as the main source file, which implicitly adds the builtin defines etc.
const MacroAnnotations & getMacroAnnotations(const IdentifierInfo *II) const
IdentifierInfo * getIdentifierInfo(StringRef Name) const
Return information about the specified preprocessor identifier token.
SourceManager & getSourceManager() const
bool isBacktrackEnabled() const
True if EnableBacktrackAtThisPos() was called and caching of tokens is on.
MacroDefinition getMacroDefinition(const IdentifierInfo *II)
bool isPreprocessedModuleFile() const
Whether the main file is preprocessed module file.
void SetPoisonReason(IdentifierInfo *II, unsigned DiagID)
Specifies the reason for poisoning an identifier.
SourceLocation CheckEndOfDirective(StringRef DirType, bool EnableMacros=false, SmallVectorImpl< Token > *ExtraToks=nullptr)
Ensure that the next token is a tok::eod token.
bool getCommentRetentionState() const
Module * getCurrentModuleImplementation()
Retrieves the module whose implementation we're current compiling, if any.
void createPreprocessingRecord()
Create a new preprocessing record, which will keep track of all macro expansions, macro definitions,...
SourceLocation SplitToken(SourceLocation TokLoc, unsigned Length)
Split the first Length characters out of the token starting at TokLoc and return a location pointing ...
Module * getCurrentModule()
Retrieves the module that we're currently building, if any.
void makeModuleVisible(Module *M, SourceLocation Loc, bool IncludeExports=true)
bool hadModuleLoaderFatalFailure() const
void setCurrentFPEvalMethod(SourceLocation PragmaLoc, LangOptions::FPEvalMethodKind Val)
bool HandleModuleContextualKeyword(Token &Result)
Callback invoked when the lexer sees one of export, import or module token at the start of a line.
const TargetInfo & getTargetInfo() const
bool LexHeaderName(Token &Result, bool AllowMacroExpansion=true)
Lex a token, forming a header-name token if possible.
bool isPCHThroughHeader(const FileEntry *FE)
Returns true if the FileEntry is the PCH through header.
void DumpLocation(SourceLocation Loc) const
bool parseSimpleIntegerLiteral(Token &Tok, uint64_t &Value)
Parses a simple integer literal to get its numeric value.
void LexUnexpandedToken(Token &Result)
Just like Lex, but disables macro expansion of identifier tokens.
bool creatingPCHWithPragmaHdrStop()
True if creating a PCH with a pragma hdrstop.
void Initialize(const TargetInfo &Target, const TargetInfo *AuxTarget=nullptr)
Initialize the preprocessor using information about the target.
FileID getPredefinesFileID() const
Returns the FileID for the preprocessor predefines.
llvm::BumpPtrAllocator & getPreprocessorAllocator()
StringRef getSpelling(SourceLocation loc, SmallVectorImpl< char > &buffer, bool *invalid=nullptr) const
Return the 'spelling' of the token at the given location; does not go up to the spelling location or ...
bool HandleComment(Token &result, SourceRange Comment)
HeaderSearch & getHeaderSearchInfo() const
bool setDeserializedSafeBufferOptOutMap(const SmallVectorImpl< SourceLocation > &SrcLocSeqs)
ExternalPreprocessorSource * getExternalSource() const
void HandleDirective(Token &Result)
Callback invoked when the lexer sees a # token at the start of a line.
SmallVector< SourceLocation, 64 > serializeSafeBufferOptOutMap() const
void recomputeCurLexerKind()
Recompute the current lexer kind based on the CurLexer/ CurTokenLexer pointers.
OptionalFileEntryRef LookupFile(SourceLocation FilenameLoc, StringRef Filename, bool isAngled, ConstSearchDirIterator FromDir, const FileEntry *FromFile, ConstSearchDirIterator *CurDir, SmallVectorImpl< char > *SearchPath, SmallVectorImpl< char > *RelativePath, ModuleMap::KnownHeader *SuggestedModule, bool *IsMapped, bool *IsFrameworkFound, bool SkipCache=false, bool OpenFile=true, bool CacheFailures=true)
Given a "foo" or <foo> reference, look up the indicated file.
IdentifierTable & getIdentifierTable()
bool LexModuleNameContinue(Token &Tok, SourceLocation UseLoc, SmallVectorImpl< Token > &Suffix, SmallVectorImpl< IdentifierLoc > &Path, bool AllowMacroExpansion, bool IsPartition)
const LangOptions & getLangOpts() const
void setTUFPEvalMethod(LangOptions::FPEvalMethodKind Val)
void CodeCompleteIncludedFile(llvm::StringRef Dir, bool IsAngled)
Hook used by the lexer to invoke the "included file" code completion point.
llvm::DenseMap< FileID, SafeBufferOptOutRegionsTy > LoadedRegions
void PoisonSEHIdentifiers(bool Poison=true)
size_t getTotalMemory() const
void LexTokensUntilEOF(std::vector< Token > *Tokens=nullptr)
Lex all tokens for this preprocessor until (and excluding) end of file.
bool isNextPPTokenOneOf(Ts... Ks) const
isNextPPTokenOneOf - Check whether the next pp-token is one of the specificed token kind.
bool usingPCHWithPragmaHdrStop()
True if using a PCH with a pragma hdrstop.
void CodeCompleteNaturalLanguage()
Hook used by the lexer to invoke the "natural language" code completion point.
void EndSourceFile()
Inform the preprocessor callbacks that processing is complete.
void CollectPPImportSuffix(SmallVectorImpl< Token > &Toks, bool StopUntilEOD=false)
Collect the tokens of a C++20 pp-import-suffix.
DiagnosticsEngine & getDiagnostics() const
bool hasSeenNoTrivialPPDirective() const
Whether we've seen pp-directives which may have changed the preprocessing state.
StringRef getLastMacroWithSpelling(SourceLocation Loc, ArrayRef< TokenValue > Tokens) const
Return the name of the macro defined before Loc that has spelling Tokens.
void setCodeCompletionIdentifierInfo(IdentifierInfo *Filter)
Set the code completion token for filtering purposes.
bool HandleModuleName(StringRef DirType, SourceLocation UseLoc, Token &Tok, SmallVectorImpl< IdentifierLoc > &Path, SmallVectorImpl< Token > &DirToks, bool AllowMacroExpansion, bool IsPartition)
DiagnosticBuilder Diag(SourceLocation Loc, unsigned DiagID) const
Forwarding function for diagnostics.
void SkipTokensWhileUsingPCH()
Skip tokens until after the include of the through header or until after a pragma hdrstop.
bool usingPCHWithThroughHeader()
True if using a PCH with a through header.
bool CollectPPImportSuffixAndEnterStream(SmallVectorImpl< Token > &Toks, bool StopUntilEOD=false)
Preprocessor(const PreprocessorOptions &PPOpts, DiagnosticsEngine &diags, const LangOptions &LangOpts, SourceManager &SM, HeaderSearch &Headers, ModuleLoader &TheModuleLoader, IdentifierInfoLookup *IILookup=nullptr, bool OwnsHeaderSearch=false, TranslationUnitKind TUKind=TU_Complete)
Represents an unpacked "presumed" location which can be presented to the user.
unsigned getColumn() const
Return the presumed column number of this location.
unsigned getLine() const
Return the presumed line number of this location.
bool isInvalid() const
Return true if this object is invalid or uninitialized.
ScratchBuffer - This class exposes a simple interface for the dynamic construction of tokens.
Encodes a location in the source.
bool isValid() const
Return true if this is a valid SourceLocation object.
void print(raw_ostream &OS, const SourceManager &SM) const
SourceLocation getLocWithOffset(IntTy Offset) const
Return a source location with the specified offset from this SourceLocation.
This class handles loading and caching of source files into memory.
std::optional< StringRef > getBufferDataOrNone(FileID FID) const
Return a StringRef to the source buffer data for the specified FileID, returning std::nullopt if inva...
A trivial tuple used to represent a source range.
StringLiteralParser - This decodes string escape characters and performs wide string analysis and Tra...
Exposes information about the current target.
Definition TargetInfo.h:226
Token - This structure provides full information about a lexed token.
Definition Token.h:36
IdentifierInfo * getIdentifierInfo() const
Definition Token.h:197
bool hasUCN() const
Returns true if this token contains a universal character name.
Definition Token.h:324
SourceLocation getLocation() const
Return a source location identifier for the specified offset in the current file.
Definition Token.h:142
unsigned getLength() const
Definition Token.h:145
bool isExpandDisabled() const
Return true if this identifier token should never be expanded in the future, due to C99 6....
Definition Token.h:298
void setKind(tok::TokenKind K)
Definition Token.h:100
bool is(tok::TokenKind K) const
is/isNot - Predicates to check if this token is a specific kind, as in "if (Tok.is(tok::l_brace)) {....
Definition Token.h:104
bool isAtStartOfLine() const
isAtStartOfLine - Return true if this token is at the start of a line.
Definition Token.h:286
bool isOneOf(Ts... Ks) const
Definition Token.h:105
@ DisableExpand
Definition Token.h:79
@ HasSeenNoTrivialPPDirective
Definition Token.h:92
@ IsReinjected
Definition Token.h:89
@ LeadingEmptyMacro
Definition Token.h:81
@ LeadingSpace
Definition Token.h:77
@ StartOfLine
Definition Token.h:75
bool isModuleContextualKeyword(bool AllowExport=true) const
Return true if we have a C++20 modules contextual keyword(export, importor module).
Definition Lexer.cpp:77
bool hasLeadingSpace() const
Return true if this token has whitespace before it.
Definition Token.h:294
bool hasLeadingEmptyMacro() const
Return true if this token has an empty macro before it.
Definition Token.h:317
bool isNot(tok::TokenKind K) const
Definition Token.h:111
void startToken()
Reset all flags to cleared.
Definition Token.h:187
bool needsCleaning() const
Return true if this token has trigraphs or escaped newlines in it.
Definition Token.h:313
void setIdentifierInfo(IdentifierInfo *II)
Definition Token.h:206
void setFlagValue(TokenFlags Flag, bool Val)
Set a flag to either true or false.
Definition Token.h:277
StringRef getRawIdentifier() const
getRawIdentifier - For a raw identifier token (i.e., an identifier lexed in raw mode),...
Definition Token.h:223
void setFlag(TokenFlags Flag)
Set the specified flag.
Definition Token.h:254
Defines the clang::TargetInfo interface.
CharacteristicKind
Indicates whether a file or directory holds normal user code, system code, or system code which is im...
const char * getTokenName(TokenKind Kind) LLVM_READNONE
Determines the name of a token as used within the front end.
const char * getKeywordSpelling(TokenKind Kind) LLVM_READNONE
Determines the spelling of simple keyword and contextual keyword tokens like 'int' and 'dynamic_cast'...
Top level wrappers for InstallAPI frontend operations.
CustomizableOptional< FileEntryRef > OptionalFileEntryRef
Definition FileEntry.h:196
@ CPlusPlus20
llvm::Registry< PragmaHandler > PragmaHandlerRegistry
Registry of pragma handlers added by plugins.
void expandUCNs(SmallVectorImpl< char > &Buf, StringRef Input)
Copy characters from Input to Buf, expanding any UCNs.
ArrayRef< IdentifierLoc > ModuleIdPath
A sequence of identifier/location pairs used to describe a particular module or submodule,...
std::pair< FileID, unsigned > FileIDAndOffset
nullptr
This class represents a compute construct, representing a 'Kind' of ‘parallel’, 'serial',...
bool isPreprocessedModuleFile(StringRef Source)
Scan an input source buffer, and check whether the input source is a preprocessed output.
@ Result
The result type of a method or function.
Definition TypeBase.h:906
ModuleUnitKind
Describes how a source input starts a C++20 module unit.
TranslationUnitKind
Describes the kind of translation unit being processed.
@ TU_Prefix
The translation unit is a prefix to a translation unit, and is not complete.
ModuleUnitKind scanInputForCXX20ModuleUnit(StringRef Source)
Scan an input source buffer to determine whether it starts a C++20 module unit, and whether that modu...
#define true
Definition stdbool.h:25