clang 24.0.0git
UnwrappedLineParser.cpp
Go to the documentation of this file.
1//===--- UnwrappedLineParser.cpp - Format C++ code ------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8///
9/// \file
10/// This file contains the implementation of the UnwrappedLineParser,
11/// which turns a stream of tokens into UnwrappedLines.
12///
13//===----------------------------------------------------------------------===//
14
15#include "UnwrappedLineParser.h"
16#include "FormatToken.h"
17#include "FormatTokenSource.h"
18#include "Macros.h"
19#include "TokenAnnotator.h"
21#include "llvm/ADT/STLExtras.h"
22#include "llvm/ADT/StringRef.h"
23#include "llvm/Support/Debug.h"
24#include "llvm/Support/raw_os_ostream.h"
25#include "llvm/Support/raw_ostream.h"
26
27#include <utility>
28
29#define DEBUG_TYPE "format-parser"
30
31namespace clang {
32namespace format {
33
34namespace {
35
36void printLine(llvm::raw_ostream &OS, const UnwrappedLine &Line,
37 StringRef Prefix = "", bool PrintText = false) {
38 OS << Prefix << "Line(" << Line.Level << ", FSC=" << Line.FirstStartColumn
39 << ")" << (Line.InPPDirective ? " MACRO" : "") << ": ";
40 bool NewLine = false;
41 for (std::list<UnwrappedLineNode>::const_iterator I = Line.Tokens.begin(),
42 E = Line.Tokens.end();
43 I != E; ++I) {
44 if (NewLine) {
45 OS << Prefix;
46 NewLine = false;
47 }
48 OS << I->Tok->Tok.getName() << "["
49 << "T=" << (unsigned)I->Tok->getType()
50 << ", OC=" << I->Tok->OriginalColumn << ", \"" << I->Tok->TokenText
51 << "\"] ";
52 for (const auto *CI = I->Children.begin(), *CE = I->Children.end();
53 CI != CE; ++CI) {
54 OS << "\n";
55 printLine(OS, *CI, (Prefix + " ").str());
56 NewLine = true;
57 }
58 }
59 if (!NewLine)
60 OS << "\n";
61}
62
63[[maybe_unused]] static void printDebugInfo(const UnwrappedLine &Line) {
64 printLine(llvm::dbgs(), Line);
65}
66
67class ScopedDeclarationState {
68public:
69 ScopedDeclarationState(UnwrappedLine &Line, llvm::BitVector &Stack,
70 bool MustBeDeclaration)
71 : Line(Line), Stack(Stack) {
72 Line.MustBeDeclaration = MustBeDeclaration;
73 Stack.push_back(MustBeDeclaration);
74 }
75 ~ScopedDeclarationState() {
76 Stack.pop_back();
77 if (!Stack.empty())
78 Line.MustBeDeclaration = Stack.back();
79 else
80 Line.MustBeDeclaration = true;
81 }
82
83private:
84 UnwrappedLine &Line;
85 llvm::BitVector &Stack;
86};
87
88} // end anonymous namespace
89
90std::ostream &operator<<(std::ostream &Stream, const UnwrappedLine &Line) {
91 llvm::raw_os_ostream OS(Stream);
92 printLine(OS, Line);
93 return Stream;
94}
95
97public:
98 // With \c DiscardLines, the lines added while in scope are discarded.
100 bool SwitchToPreprocessorLines = false,
101 bool DiscardLines = false)
102 : Parser(Parser), OriginalLines(Parser.CurrentLines),
103 DiscardLines(DiscardLines) {
104 if (SwitchToPreprocessorLines)
105 Parser.CurrentLines = &Parser.PreprocessorDirectives;
106 else if (!Parser.Line->Tokens.empty())
107 Parser.CurrentLines = &Parser.Line->Tokens.back().Children;
108 OriginalNumLines = Parser.CurrentLines->size();
109 PreBlockLine = std::move(Parser.Line);
110 Parser.Line = std::make_unique<UnwrappedLine>();
111 Parser.Line->Level = PreBlockLine->Level;
112 Parser.Line->PPLevel = PreBlockLine->PPLevel;
113 Parser.Line->InPPDirective = PreBlockLine->InPPDirective;
114 Parser.Line->InMacroBody = PreBlockLine->InMacroBody;
115 Parser.Line->UnbracedBodyLevel = PreBlockLine->UnbracedBodyLevel;
116 }
117
119 if (!Parser.Line->Tokens.empty())
120 Parser.addUnwrappedLine();
121 assert(Parser.Line->Tokens.empty());
122 if (DiscardLines)
123 Parser.CurrentLines->truncate(OriginalNumLines);
124 Parser.Line = std::move(PreBlockLine);
125 if (Parser.CurrentLines == &Parser.PreprocessorDirectives)
126 Parser.PP.AtEndOfPPLine = true;
127 Parser.CurrentLines = OriginalLines;
128 }
129
130private:
132
133 std::unique_ptr<UnwrappedLine> PreBlockLine;
134 SmallVectorImpl<UnwrappedLine> *OriginalLines;
135 size_t OriginalNumLines;
136 bool DiscardLines;
137};
138
140public:
142 const FormatStyle &Style, unsigned &LineLevel)
144 Style.BraceWrapping.AfterControlStatement ==
145 FormatStyle::BWACS_Always,
146 Style.BraceWrapping.IndentBraces) {}
148 bool WrapBrace, bool IndentBrace)
149 : LineLevel(LineLevel), OldLineLevel(LineLevel) {
150 if (WrapBrace)
151 Parser->addUnwrappedLine();
152 if (IndentBrace)
153 ++LineLevel;
154 }
155 ~CompoundStatementIndenter() { LineLevel = OldLineLevel; }
156
157private:
158 unsigned &LineLevel;
159 unsigned OldLineLevel;
160};
161
163 SourceManager &SourceMgr, const FormatStyle &Style,
164 const AdditionalKeywords &Keywords, unsigned FirstStartColumn,
166 llvm::SpecificBumpPtrAllocator<FormatToken> &Allocator,
167 IdentifierTable &IdentTable)
168 : Line(new UnwrappedLine), CurrentLines(&Lines), Style(Style),
169 IsCpp(Style.isCpp()), LangOpts(getFormattingLangOpts(Style)),
170 Keywords(Keywords), CommentPragmasRegex(Style.CommentPragmas),
171 Tokens(nullptr), Callback(Callback), AllTokens(Tokens),
172 PP(getIncludeGuardState(Style.IndentPPDirectives)),
173 FirstStartColumn(FirstStartColumn),
174 Macros(Style.Macros, SourceMgr, Style, Allocator, IdentTable) {}
175
176void UnwrappedLineParser::reset() {
177 PP.BranchLevel = -1;
178 PP.IncludeGuard = getIncludeGuardState(Style.IndentPPDirectives);
179 PP.IncludeGuardToken = nullptr;
180 ParsedPPDirectives.clear();
181 Line.reset(new UnwrappedLine);
182 CommentsBeforeNextToken.clear();
183 FormatTok = nullptr;
184 PP.AtEndOfPPLine = false;
185 IsDecltypeAutoFunction = false;
186 PreprocessorDirectives.clear();
187 CurrentLines = &Lines;
188 DeclarationScopeStack.clear();
189 NestedTooDeep.clear();
190 NestedLambdas.clear();
191 PP.Stack.clear();
192 Line->FirstStartColumn = FirstStartColumn;
193
194 if (!Unexpanded.empty())
195 for (FormatToken *Token : AllTokens)
196 Token->MacroCtx.reset();
197 CurrentExpandedLines.clear();
198 ExpandedLines.clear();
199 Unexpanded.clear();
200 InExpansion = false;
201 Reconstruct.reset();
202}
203
205 IndexedTokenSource TokenSource(AllTokens);
206 Line->FirstStartColumn = FirstStartColumn;
207 do {
208 LLVM_DEBUG(llvm::dbgs() << "----\n");
209 reset();
210 Tokens = &TokenSource;
211 TokenSource.reset();
212
213 readToken();
214 parseFile();
215
216 // If we found an include guard then all preprocessor directives (other than
217 // the guard) are over-indented by one.
218 if (PP.IncludeGuard == IG_Found) {
219 for (auto &Line : Lines)
220 if (Line.InPPDirective && Line.Level > 0)
221 --Line.Level;
222 }
223
224 // Create line with eof token.
225 assert(eof());
226 pushToken(FormatTok);
227 addUnwrappedLine();
228
229 // In a first run, format everything with the lines containing macro calls
230 // replaced by the expansion.
231 if (!ExpandedLines.empty()) {
232 LLVM_DEBUG(llvm::dbgs() << "Expanded lines:\n");
233 for (const auto &Line : Lines) {
234 if (!Line.Tokens.empty()) {
235 auto it = ExpandedLines.find(Line.Tokens.begin()->Tok);
236 if (it != ExpandedLines.end()) {
237 for (const auto &Expanded : it->second) {
238 LLVM_DEBUG(printDebugInfo(Expanded));
239 Callback.consumeUnwrappedLine(Expanded);
240 }
241 continue;
242 }
243 }
244 LLVM_DEBUG(printDebugInfo(Line));
245 Callback.consumeUnwrappedLine(Line);
246 }
247 Callback.finishRun();
248 }
249
250 LLVM_DEBUG(llvm::dbgs() << "Unwrapped lines:\n");
251 for (const UnwrappedLine &Line : Lines) {
252 LLVM_DEBUG(printDebugInfo(Line));
253 Callback.consumeUnwrappedLine(Line);
254 }
255 Callback.finishRun();
256 Lines.clear();
257 while (!PP.LevelBranchIndex.empty() &&
258 PP.LevelBranchIndex.back() + 1 >= PP.LevelBranchCount.back()) {
259 PP.LevelBranchIndex.resize(PP.LevelBranchIndex.size() - 1);
260 PP.LevelBranchCount.resize(PP.LevelBranchCount.size() - 1);
261 }
262 if (!PP.LevelBranchIndex.empty()) {
263 ++PP.LevelBranchIndex.back();
264 assert(PP.LevelBranchIndex.size() == PP.LevelBranchCount.size());
265 assert(PP.LevelBranchIndex.back() <= PP.LevelBranchCount.back());
266 }
267 } while (!PP.LevelBranchIndex.empty());
268}
269
270void UnwrappedLineParser::parseFile() {
271 // The top-level context in a file always has declarations, except for pre-
272 // processor directives and JavaScript files.
273 bool MustBeDeclaration = !Line->InPPDirective && !Style.isJavaScript();
274 ScopedDeclarationState DeclarationState(*Line, DeclarationScopeStack,
275 MustBeDeclaration);
276 if (Style.isTextProto() || (Style.isJson() && FormatTok->IsFirst))
277 parseBracedList();
278 else
279 parseLevel();
280 // Make sure to format the remaining tokens.
281 //
282 // LK_TextProto is special since its top-level is parsed as the body of a
283 // braced list, which does not necessarily have natural line separators such
284 // as a semicolon. Comments after the last entry that have been determined to
285 // not belong to that line, as in:
286 // key: value
287 // // endfile comment
288 // do not have a chance to be put on a line of their own until this point.
289 // Here we add this newline before end-of-file comments.
290 if (Style.isTextProto() && !CommentsBeforeNextToken.empty())
291 addUnwrappedLine();
292 flushComments(true);
293 addUnwrappedLine();
294}
295
296void UnwrappedLineParser::parseCSharpGenericTypeConstraint() {
297 do {
298 switch (FormatTok->Tok.getKind()) {
299 case tok::l_brace:
300 case tok::semi:
301 return;
302 default:
303 if (FormatTok->is(Keywords.kw_where)) {
304 addUnwrappedLine();
305 nextToken();
306 parseCSharpGenericTypeConstraint();
307 break;
308 }
309 nextToken();
310 break;
311 }
312 } while (!eof());
313}
314
315void UnwrappedLineParser::parseCSharpAttribute() {
316 int UnpairedSquareBrackets = 1;
317 do {
318 switch (FormatTok->Tok.getKind()) {
319 case tok::r_square:
320 nextToken();
321 --UnpairedSquareBrackets;
322 if (UnpairedSquareBrackets == 0) {
323 addUnwrappedLine();
324 return;
325 }
326 break;
327 case tok::l_square:
328 ++UnpairedSquareBrackets;
329 nextToken();
330 break;
331 default:
332 nextToken();
333 break;
334 }
335 } while (!eof());
336}
337
338bool UnwrappedLineParser::precededByCommentOrPPDirective() const {
339 if (!Lines.empty() && Lines.back().InPPDirective)
340 return true;
341
342 const FormatToken *Previous = Tokens->getPreviousToken();
343 return Previous && Previous->is(tok::comment) &&
344 (Previous->IsMultiline || Previous->NewlinesBefore > 0);
345}
346
347/// Parses a level, that is ???.
348/// \param OpeningBrace Opening brace (\p nullptr if absent) of that level.
349/// \param IfKind The \p if statement kind in the level.
350/// \param IfLeftBrace The left brace of the \p if block in the level.
351/// \returns true if a simple block of if/else/for/while, or false otherwise.
352/// (A simple block has a single statement.)
353bool UnwrappedLineParser::parseLevel(const FormatToken *OpeningBrace,
354 IfStmtKind *IfKind,
355 FormatToken **IfLeftBrace,
356 bool *SeenExplicitAccessModifier) {
357 const bool InRequiresExpression =
358 OpeningBrace && OpeningBrace->is(TT_RequiresExpressionLBrace);
359 const bool IsPrecededByCommentOrPPDirective =
360 !Style.RemoveBracesLLVM || precededByCommentOrPPDirective();
361 FormatToken *IfLBrace = nullptr;
362 bool HasDoWhile = false;
363 bool HasLabel = false;
364 unsigned StatementCount = 0;
365 bool SwitchLabelEncountered = false;
366
367 do {
368 if (FormatTok->isAttribute()) {
369 nextToken();
370 if (FormatTok->is(tok::l_paren))
371 parseParens();
372 continue;
373 }
374 tok::TokenKind Kind = FormatTok->Tok.getKind();
375 if (FormatTok->is(TT_MacroBlockBegin))
376 Kind = tok::l_brace;
377 else if (FormatTok->is(TT_MacroBlockEnd))
378 Kind = tok::r_brace;
379
380 auto ParseDefault = [this, OpeningBrace, IfKind, &IfLBrace, &HasDoWhile,
381 &HasLabel, &StatementCount,
382 SeenExplicitAccessModifier] {
383 if (SeenExplicitAccessModifier && !*SeenExplicitAccessModifier) {
384 const bool IsQtAccessLabel =
385 FormatTok->isOneOf(Keywords.kw_signals, Keywords.kw_qsignals,
386 Keywords.kw_slots, Keywords.kw_qslots) &&
387 Tokens->peekNextToken(/*SkipComment=*/true)->is(tok::colon);
388 if (FormatTok->isAccessSpecifierKeyword() || IsQtAccessLabel) {
389 ++Line->Level;
390 *SeenExplicitAccessModifier = true;
391 }
392 }
393 parseStructuralElement(OpeningBrace, IfKind, &IfLBrace,
394 HasDoWhile ? nullptr : &HasDoWhile,
395 HasLabel ? nullptr : &HasLabel);
396 ++StatementCount;
397 assert(StatementCount > 0 && "StatementCount overflow!");
398 };
399
400 switch (Kind) {
401 case tok::comment:
402 nextToken();
403 addUnwrappedLine();
404 break;
405 case tok::l_brace:
406 if (InRequiresExpression) {
407 FormatTok->setFinalizedType(TT_CompoundRequirementLBrace);
408 } else if (FormatTok->Previous &&
409 FormatTok->Previous->ClosesRequiresClause) {
410 // We need the 'default' case here to correctly parse a function
411 // l_brace.
412 ParseDefault();
413 continue;
414 }
415 if (!InRequiresExpression && FormatTok->isNot(TT_MacroBlockBegin)) {
416 if (tryToParseBracedList())
417 continue;
418 FormatTok->setFinalizedType(TT_BlockLBrace);
419 }
420 parseBlock();
421 ++StatementCount;
422 assert(StatementCount > 0 && "StatementCount overflow!");
423 addUnwrappedLine();
424 break;
425 case tok::r_brace:
426 if (OpeningBrace) {
427 if (!Style.RemoveBracesLLVM || Line->InPPDirective ||
428 OpeningBrace->isNoneOf(TT_ControlStatementLBrace, TT_ElseLBrace)) {
429 return false;
430 }
431 if (FormatTok->isNot(tok::r_brace) || StatementCount != 1 || HasLabel ||
432 HasDoWhile || IsPrecededByCommentOrPPDirective ||
433 precededByCommentOrPPDirective()) {
434 return false;
435 }
436 const FormatToken *Next = Tokens->peekNextToken();
437 if (Next->is(tok::comment) && Next->NewlinesBefore == 0)
438 return false;
439 if (IfLeftBrace)
440 *IfLeftBrace = IfLBrace;
441 return true;
442 }
443 nextToken();
444 addUnwrappedLine();
445 break;
446 case tok::kw_default: {
447 unsigned StoredPosition = Tokens->getPosition();
448 auto *Next = Tokens->getNextNonComment();
449 FormatTok = Tokens->setPosition(StoredPosition);
450 if (Next->isNoneOf(tok::colon, tok::arrow)) {
451 // default not followed by `:` or `->` is not a case label; treat it
452 // like an identifier.
453 parseStructuralElement();
454 break;
455 }
456 // Else, if it is 'default:', fall through to the case handling.
457 [[fallthrough]];
458 }
459 case tok::kw_case:
460 if (Style.Language == FormatStyle::LK_Proto || Style.isVerilog() ||
461 (Style.isJavaScript() && Line->MustBeDeclaration)) {
462 // Proto: there are no switch/case statements
463 // Verilog: Case labels don't have this word. We handle case
464 // labels including default in TokenAnnotator.
465 // JavaScript: A 'case: string' style field declaration.
466 ParseDefault();
467 break;
468 }
469 if (!SwitchLabelEncountered &&
470 (Style.IndentCaseLabels ||
471 (OpeningBrace && OpeningBrace->is(TT_SwitchExpressionLBrace)) ||
472 (Line->InPPDirective && Line->Level == 1))) {
473 ++Line->Level;
474 }
475 SwitchLabelEncountered = true;
476 parseStructuralElement();
477 break;
478 case tok::l_square:
479 if (Style.isCSharp()) {
480 nextToken();
481 parseCSharpAttribute();
482 break;
483 }
484 if (handleCppAttributes())
485 break;
486 [[fallthrough]];
487 default:
488 ParseDefault();
489 break;
490 }
491 } while (!eof());
492
493 return false;
494}
495
496void UnwrappedLineParser::calculateBraceTypes(bool ExpectClassBody) {
497 // We'll parse forward through the tokens until we hit
498 // a closing brace or eof - note that getNextToken() will
499 // parse macros, so this will magically work inside macro
500 // definitions, too.
501 unsigned StoredPosition = Tokens->getPosition();
502 FormatToken *Tok = FormatTok;
503 const FormatToken *PrevTok = Tok->Previous;
504 // Keep a stack of positions of lbrace tokens. We will
505 // update information about whether an lbrace starts a
506 // braced init list or a different block during the loop.
507 struct StackEntry {
509 const FormatToken *PrevTok;
510 };
511 SmallVector<StackEntry, 8> LBraceStack;
512 assert(Tok->is(tok::l_brace));
513
514 do {
515 auto *NextTok = Tokens->getNextNonComment();
516
517 if (!Line->InMacroBody && !Style.isTableGen()) {
518 // Skip PPDirective lines (except macro definitions) and comments.
519 while (NextTok->is(tok::hash)) {
520 NextTok = Tokens->getNextToken();
521 if (NextTok->isOneOf(tok::pp_not_keyword, tok::pp_define))
522 break;
523 do {
524 NextTok = Tokens->getNextToken();
525 } while (!NextTok->HasUnescapedNewline && NextTok->isNot(tok::eof));
526
527 while (NextTok->is(tok::comment))
528 NextTok = Tokens->getNextToken();
529 }
530 }
531
532 switch (Tok->Tok.getKind()) {
533 case tok::l_brace:
534 if (Style.isJavaScript() && PrevTok) {
535 if (PrevTok->isOneOf(tok::colon, tok::less)) {
536 // A ':' indicates this code is in a type, or a braced list
537 // following a label in an object literal ({a: {b: 1}}).
538 // A '<' could be an object used in a comparison, but that is nonsense
539 // code (can never return true), so more likely it is a generic type
540 // argument (`X<{a: string; b: number}>`).
541 // The code below could be confused by semicolons between the
542 // individual members in a type member list, which would normally
543 // trigger BK_Block. In both cases, this must be parsed as an inline
544 // braced init.
545 Tok->setBlockKind(BK_BracedInit);
546 } else if (PrevTok->is(tok::r_paren)) {
547 // `) { }` can only occur in function or method declarations in JS.
548 Tok->setBlockKind(BK_Block);
549 }
550 } else if (Style.isJava() && PrevTok && PrevTok->is(tok::arrow)) {
551 Tok->setBlockKind(BK_Block);
552 } else {
553 Tok->setBlockKind(BK_Unknown);
554 }
555 LBraceStack.push_back({Tok, PrevTok});
556 break;
557 case tok::r_brace:
558 if (LBraceStack.empty())
559 break;
560 if (auto *LBrace = LBraceStack.back().Tok; LBrace->is(BK_Unknown)) {
561 bool ProbablyBracedList = false;
562 if (Style.Language == FormatStyle::LK_Proto) {
563 ProbablyBracedList = NextTok->isOneOf(tok::comma, tok::r_square);
564 } else if (LBrace->isNot(TT_EnumLBrace)) {
565 // Using OriginalColumn to distinguish between ObjC methods and
566 // binary operators is a bit hacky.
567 bool NextIsObjCMethod = NextTok->isOneOf(tok::plus, tok::minus) &&
568 NextTok->OriginalColumn == 0;
569
570 // Try to detect a braced list. Note that regardless how we mark inner
571 // braces here, we will overwrite the BlockKind later if we parse a
572 // braced list (where all blocks inside are by default braced lists),
573 // or when we explicitly detect blocks (for example while parsing
574 // lambdas).
575
576 // If we already marked the opening brace as braced list, the closing
577 // must also be part of it.
578 ProbablyBracedList = LBrace->is(TT_BracedListLBrace);
579
580 ProbablyBracedList = ProbablyBracedList ||
581 (Style.isJavaScript() &&
582 NextTok->isOneOf(Keywords.kw_of, Keywords.kw_in,
583 Keywords.kw_as));
584 ProbablyBracedList =
585 ProbablyBracedList ||
586 (IsCpp && (PrevTok->Tok.isLiteral() ||
587 NextTok->isOneOf(tok::l_paren, tok::arrow)));
588
589 // If there is a comma, or right paren after the closing brace, we
590 // assume this is a braced initializer list.
591 // FIXME: Some of these do not apply to JS, e.g. "} {" can never be a
592 // braced list in JS.
593 ProbablyBracedList =
594 ProbablyBracedList ||
595 NextTok->isOneOf(tok::comma, tok::period, tok::colon,
596 tok::r_paren, tok::r_square, tok::ellipsis);
597
598 // Distinguish between braced list in a constructor initializer list
599 // followed by constructor body, or just adjacent blocks.
600 ProbablyBracedList =
601 ProbablyBracedList ||
602 (NextTok->is(tok::l_brace) && LBraceStack.back().PrevTok &&
603 LBraceStack.back().PrevTok->isOneOf(tok::identifier,
604 tok::greater));
605
606 ProbablyBracedList =
607 ProbablyBracedList ||
608 (NextTok->is(tok::identifier) &&
609 PrevTok->isNoneOf(tok::semi, tok::r_brace, tok::l_brace));
610
611 ProbablyBracedList = ProbablyBracedList ||
612 (NextTok->is(tok::semi) &&
613 (!ExpectClassBody || LBraceStack.size() != 1));
614
615 ProbablyBracedList =
616 ProbablyBracedList ||
617 (NextTok->isBinaryOperator() && !NextIsObjCMethod);
618
619 if (!Style.isCSharp() && NextTok->is(tok::l_square)) {
620 // We can have an array subscript after a braced init
621 // list, but C++11 attributes are expected after blocks.
622 NextTok = Tokens->getNextToken();
623 ProbablyBracedList = NextTok->isNot(tok::l_square);
624 }
625
626 // Cpp macro definition body that is a nonempty braced list or block:
627 if (IsCpp && Line->InMacroBody && PrevTok != FormatTok &&
628 !FormatTok->Previous && NextTok->is(tok::eof) &&
629 // A statement can end with only `;` (simple statement), a block
630 // closing brace (compound statement), or `:` (label statement).
631 // If PrevTok is a block opening brace, Tok ends an empty block.
632 PrevTok->isNoneOf(tok::semi, BK_Block, tok::colon)) {
633 ProbablyBracedList = true;
634 }
635 }
636 const auto BlockKind = ProbablyBracedList ? BK_BracedInit : BK_Block;
637 Tok->setBlockKind(BlockKind);
638 LBrace->setBlockKind(BlockKind);
639 }
640 LBraceStack.pop_back();
641 break;
642 case tok::identifier:
643 if (Tok->isNot(TT_StatementMacro))
644 break;
645 [[fallthrough]];
646 case tok::at:
647 case tok::semi:
648 case tok::kw_if:
649 case tok::kw_while:
650 case tok::kw_for:
651 case tok::kw_switch:
652 case tok::kw_try:
653 case tok::kw___try:
654 if (!LBraceStack.empty() && LBraceStack.back().Tok->is(BK_Unknown))
655 LBraceStack.back().Tok->setBlockKind(BK_Block);
656 break;
657 default:
658 break;
659 }
660
661 PrevTok = Tok;
662 Tok = NextTok;
663 } while (Tok->isNot(tok::eof) && !LBraceStack.empty());
664
665 // Assume other blocks for all unclosed opening braces.
666 for (const auto &Entry : LBraceStack)
667 if (Entry.Tok->is(BK_Unknown))
668 Entry.Tok->setBlockKind(BK_Block);
669
670 FormatTok = Tokens->setPosition(StoredPosition);
671}
672
673// Sets the token type of the directly previous right brace.
674void UnwrappedLineParser::setPreviousRBraceType(TokenType Type) {
675 if (auto Prev = FormatTok->getPreviousNonComment();
676 Prev && Prev->is(tok::r_brace)) {
677 Prev->setFinalizedType(Type);
678 }
679}
680
681template <class T>
682static inline void hash_combine(std::size_t &seed, const T &v) {
683 std::hash<T> hasher;
684 seed ^= hasher(v) + 0x9e3779b9 + (seed << 6) + (seed >> 2);
685}
686
687size_t UnwrappedLineParser::computePPHash() const {
688 size_t h = 0;
689 for (const auto &i : PP.Stack) {
690 hash_combine(h, size_t(i.Kind));
691 hash_combine(h, i.Line);
692 }
693 return h;
694}
695
696// Checks whether \p ParsedLine might fit on a single line. If \p OpeningBrace
697// is not null, subtracts its length (plus the preceding space) when computing
698// the length of \p ParsedLine. We must clone the tokens of \p ParsedLine before
699// running the token annotator on it so that we can restore them afterward.
700bool UnwrappedLineParser::mightFitOnOneLine(
701 UnwrappedLine &ParsedLine, const FormatToken *OpeningBrace) const {
702 const auto ColumnLimit = Style.ColumnLimit;
703 if (ColumnLimit == 0)
704 return true;
705
706 auto &Tokens = ParsedLine.Tokens;
707 assert(!Tokens.empty());
708
709 const auto *LastToken = Tokens.back().Tok;
710 assert(LastToken);
711
712 SmallVector<UnwrappedLineNode> SavedTokens(Tokens.size());
713
714 int Index = 0;
715 for (const auto &Token : Tokens) {
716 assert(Token.Tok);
717 auto &SavedToken = SavedTokens[Index++];
718 SavedToken.Tok = new FormatToken;
719 SavedToken.Tok->copyFrom(*Token.Tok);
720 SavedToken.Children = std::move(Token.Children);
721 }
722
723 AnnotatedLine Line(ParsedLine);
724 assert(Line.Last == LastToken);
725
726 TokenAnnotator Annotator(Style, Keywords);
727 Annotator.annotate(Line);
728 Annotator.calculateFormattingInformation(Line);
729
730 auto Length = LastToken->TotalLength;
731 if (OpeningBrace) {
732 assert(OpeningBrace != Tokens.front().Tok);
733 if (auto Prev = OpeningBrace->Previous;
734 Prev && Prev->TotalLength + ColumnLimit == OpeningBrace->TotalLength) {
735 Length -= ColumnLimit;
736 }
737 Length -= OpeningBrace->TokenText.size() + 1;
738 }
739
740 if (const auto *FirstToken = Line.First; FirstToken->is(tok::r_brace)) {
741 assert(!OpeningBrace || OpeningBrace->is(TT_ControlStatementLBrace));
742 Length -= FirstToken->TokenText.size() + 1;
743 }
744
745 Index = 0;
746 for (auto &Token : Tokens) {
747 const auto &SavedToken = SavedTokens[Index++];
748 Token.Tok->copyFrom(*SavedToken.Tok);
749 Token.Children = std::move(SavedToken.Children);
750 delete SavedToken.Tok;
751 }
752
753 // If these change PPLevel needs to be used for get correct indentation.
754 assert(!Line.InMacroBody);
755 assert(!Line.InPPDirective);
756 return Line.Level * Style.IndentWidth + Length <= ColumnLimit;
757}
758
759FormatToken *UnwrappedLineParser::parseBlock(
760 bool MustBeDeclaration, unsigned AddLevels, bool MunchSemi, bool KeepBraces,
761 IfStmtKind *IfKind, bool UnindentWhitesmithsBraces,
762 bool IndentAfterExplicitAccessModifier) {
763 auto HandleVerilogBlockLabel = [this]() {
764 // ":" name
765 if (Style.isVerilog() && FormatTok->is(tok::colon)) {
766 nextToken();
767 if (Keywords.isVerilogIdentifier(*FormatTok))
768 nextToken();
769 }
770 };
771
772 // Whether this is a Verilog-specific block that has a special header like a
773 // module.
774 const bool VerilogHierarchy =
775 Style.isVerilog() && Keywords.isVerilogHierarchy(*FormatTok);
776 assert((FormatTok->isOneOf(tok::l_brace, TT_MacroBlockBegin) ||
777 (Style.isVerilog() &&
778 (Keywords.isVerilogBegin(*FormatTok) || VerilogHierarchy))) &&
779 "'{' or macro block token expected");
780 FormatToken *Tok = FormatTok;
781 const bool FollowedByComment = Tokens->peekNextToken()->is(tok::comment);
782 auto Index = CurrentLines->size();
783 const bool MacroBlock = FormatTok->is(TT_MacroBlockBegin);
784 FormatTok->setBlockKind(BK_Block);
785
786 const bool IsWhitesmiths =
787 Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths;
788
789 // For Whitesmiths mode, jump to the next level prior to skipping over the
790 // braces.
791 if (!VerilogHierarchy && AddLevels > 0 && IsWhitesmiths)
792 ++Line->Level;
793
794 size_t PPStartHash = computePPHash();
795
796 const unsigned InitialLevel = Line->Level;
797 if (VerilogHierarchy) {
798 AddLevels += parseVerilogHierarchyHeader();
799 } else {
800 nextToken(/*LevelDifference=*/AddLevels);
801 HandleVerilogBlockLabel();
802 }
803
804 // Bail out if there are too many levels. Otherwise, the stack might overflow.
805 if (Line->Level > 300)
806 return nullptr;
807
808 if (MacroBlock && FormatTok->is(tok::l_paren))
809 parseParens();
810
811 size_t NbPreprocessorDirectives =
812 !parsingPPDirective() ? PreprocessorDirectives.size() : 0;
813 addUnwrappedLine();
814 size_t OpeningLineIndex =
815 CurrentLines->empty()
817 : (CurrentLines->size() - 1 - NbPreprocessorDirectives);
818
819 // Whitesmiths is weird here. The brace needs to be indented for the namespace
820 // block, but the block itself may not be indented depending on the style
821 // settings. This allows the format to back up one level in those cases.
822 if (UnindentWhitesmithsBraces)
823 --Line->Level;
824
825 ScopedDeclarationState DeclarationState(*Line, DeclarationScopeStack,
826 MustBeDeclaration);
827
828 // Whitesmiths logic has already added a level by this point, so avoid
829 // adding it twice.
830 if (AddLevels > 0u)
831 Line->Level += AddLevels - (IsWhitesmiths ? 1 : 0);
832
833 FormatToken *IfLBrace = nullptr;
834 bool SeenExplicitAccessModifier = false;
835 const bool SimpleBlock =
836 parseLevel(Tok, IfKind, &IfLBrace,
837 IndentAfterExplicitAccessModifier ? &SeenExplicitAccessModifier
838 : nullptr);
839
840 if (eof())
841 return IfLBrace;
842
843 if (MacroBlock ? FormatTok->isNot(TT_MacroBlockEnd)
844 : FormatTok->isNot(tok::r_brace)) {
845 Line->Level = InitialLevel;
846 FormatTok->setBlockKind(BK_Block);
847 return IfLBrace;
848 }
849
850 if (FormatTok->is(tok::r_brace)) {
851 FormatTok->setBlockKind(BK_Block);
852 if (Tok->is(TT_NamespaceLBrace))
853 FormatTok->setFinalizedType(TT_NamespaceRBrace);
854 }
855
856 const bool IsFunctionRBrace =
857 FormatTok->is(tok::r_brace) && Tok->is(TT_FunctionLBrace);
858
859 auto RemoveBraces = [=]() mutable {
860 if (!SimpleBlock)
861 return false;
862 assert(Tok->isOneOf(TT_ControlStatementLBrace, TT_ElseLBrace));
863 assert(FormatTok->is(tok::r_brace));
864 const bool WrappedOpeningBrace = !Tok->Previous;
865 if (WrappedOpeningBrace && FollowedByComment)
866 return false;
867 const bool HasRequiredIfBraces = IfLBrace && !IfLBrace->Optional;
868 if (KeepBraces && !HasRequiredIfBraces)
869 return false;
870 if (Tok->isNot(TT_ElseLBrace) || !HasRequiredIfBraces) {
871 const FormatToken *Previous = Tokens->getPreviousToken();
872 assert(Previous);
873 if (Previous->is(tok::r_brace) && !Previous->Optional)
874 return false;
875 }
876 assert(!CurrentLines->empty());
877 auto &LastLine = CurrentLines->back();
878 if (LastLine.Level == InitialLevel + 1 && !mightFitOnOneLine(LastLine))
879 return false;
880 if (Tok->is(TT_ElseLBrace))
881 return true;
882 if (WrappedOpeningBrace) {
883 assert(Index > 0);
884 --Index; // The line above the wrapped l_brace.
885 Tok = nullptr;
886 }
887 return mightFitOnOneLine((*CurrentLines)[Index], Tok);
888 };
889 if (RemoveBraces()) {
890 Tok->MatchingParen = FormatTok;
891 FormatTok->MatchingParen = Tok;
892 }
893
894 size_t PPEndHash = computePPHash();
895
896 if (SeenExplicitAccessModifier)
897 ++AddLevels;
898 // Munch the closing brace.
899 nextToken(/*LevelDifference=*/-AddLevels);
900
901 // When this is a function block and there is an unnecessary semicolon
902 // afterwards then mark it as optional (so the RemoveSemi pass can get rid of
903 // it later).
904 if (Style.RemoveSemicolon && IsFunctionRBrace) {
905 while (FormatTok->is(tok::semi)) {
906 FormatTok->Optional = true;
907 nextToken();
908 }
909 }
910
911 HandleVerilogBlockLabel();
912
913 if (MacroBlock && FormatTok->is(tok::l_paren))
914 parseParens();
915
916 Line->Level = InitialLevel;
917
918 if (FormatTok->is(tok::kw_noexcept)) {
919 // A noexcept in a requires expression.
920 nextToken();
921 }
922
923 if (FormatTok->is(tok::arrow)) {
924 // Following the } or noexcept we can find a trailing return type arrow
925 // as part of an implicit conversion constraint.
926 nextToken();
927 parseStructuralElement();
928 }
929
930 if (MunchSemi && FormatTok->is(tok::semi))
931 nextToken();
932
933 if (PPStartHash == PPEndHash) {
934 Line->MatchingOpeningBlockLineIndex = OpeningLineIndex;
935 if (OpeningLineIndex != UnwrappedLine::kInvalidIndex) {
936 // Update the opening line to add the forward reference as well
937 (*CurrentLines)[OpeningLineIndex].MatchingClosingBlockLineIndex =
938 CurrentLines->size() - 1;
939 }
940 }
941
942 return IfLBrace;
943}
944
945static bool isGoogScope(const UnwrappedLine &Line) {
946 // FIXME: Closure-library specific stuff should not be hard-coded but be
947 // configurable.
948 if (Line.Tokens.size() < 4)
949 return false;
950 auto I = Line.Tokens.begin();
951 if (I->Tok->TokenText != "goog")
952 return false;
953 ++I;
954 if (I->Tok->isNot(tok::period))
955 return false;
956 ++I;
957 if (I->Tok->TokenText != "scope")
958 return false;
959 ++I;
960 return I->Tok->is(tok::l_paren);
961}
962
963static bool isIIFE(const UnwrappedLine &Line,
964 const AdditionalKeywords &Keywords) {
965 // Look for the start of an immediately invoked anonymous function.
966 // https://en.wikipedia.org/wiki/Immediately-invoked_function_expression
967 // This is commonly done in JavaScript to create a new, anonymous scope.
968 // Example: (function() { ... })()
969 if (Line.Tokens.size() < 3)
970 return false;
971 auto I = Line.Tokens.begin();
972 if (I->Tok->isNot(tok::l_paren))
973 return false;
974 ++I;
975 if (I->Tok->isNot(Keywords.kw_function))
976 return false;
977 ++I;
978 return I->Tok->is(tok::l_paren);
979}
980
981static bool ShouldBreakBeforeBrace(const FormatStyle &Style,
982 const FormatToken &InitialToken,
983 bool IsEmptyBlock,
984 bool IsJavaRecord = false) {
985 if (IsJavaRecord)
986 return Style.BraceWrapping.AfterClass;
987
988 tok::TokenKind Kind = InitialToken.Tok.getKind();
989 if (InitialToken.is(TT_NamespaceMacro))
990 Kind = tok::kw_namespace;
991
992 const bool WrapRecordAllowed =
993 !IsEmptyBlock ||
994 Style.AllowShortRecordOnASingleLine < FormatStyle::SRS_Empty ||
995 Style.BraceWrapping.SplitEmptyRecord;
996
997 switch (Kind) {
998 case tok::kw_namespace:
999 return Style.BraceWrapping.AfterNamespace;
1000 case tok::kw_class:
1001 return Style.BraceWrapping.AfterClass && WrapRecordAllowed;
1002 case tok::kw_union:
1003 return Style.BraceWrapping.AfterUnion && WrapRecordAllowed;
1004 case tok::kw_struct:
1005 return Style.BraceWrapping.AfterStruct && WrapRecordAllowed;
1006 case tok::kw_enum:
1007 return Style.BraceWrapping.AfterEnum;
1008 default:
1009 return false;
1010 }
1011}
1012
1013void UnwrappedLineParser::parseChildBlock() {
1014 assert(FormatTok->is(tok::l_brace));
1015 FormatTok->setBlockKind(BK_Block);
1016 const FormatToken *OpeningBrace = FormatTok;
1017 nextToken();
1018 {
1019 bool SkipIndent = (Style.isJavaScript() &&
1020 (isGoogScope(*Line) || isIIFE(*Line, Keywords)));
1021 ScopedLineState LineState(*this);
1022 ScopedDeclarationState DeclarationState(*Line, DeclarationScopeStack,
1023 /*MustBeDeclaration=*/false);
1024 Line->Level += SkipIndent ? 0 : 1;
1025 parseLevel(OpeningBrace);
1026 flushComments(isOnNewLine(*FormatTok));
1027 Line->Level -= SkipIndent ? 0 : 1;
1028 }
1029 nextToken();
1030}
1031
1032void UnwrappedLineParser::parsePPDirective() {
1033 assert(FormatTok->is(tok::hash) && "'#' expected");
1034 ScopedMacroState MacroState(*Line, Tokens, FormatTok);
1035
1036 nextToken();
1037
1038 if (!FormatTok->Tok.getIdentifierInfo()) {
1039 parsePPUnknown();
1040 return;
1041 }
1042
1043 switch (FormatTok->Tok.getIdentifierInfo()->getPPKeywordID()) {
1044 case tok::pp_define:
1045 parsePPDefine();
1046 return;
1047 case tok::pp_if:
1048 parsePPIf(/*IfDef=*/false);
1049 break;
1050 case tok::pp_ifdef:
1051 case tok::pp_ifndef:
1052 parsePPIf(/*IfDef=*/true);
1053 break;
1054 case tok::pp_else:
1055 case tok::pp_elifdef:
1056 case tok::pp_elifndef:
1057 case tok::pp_elif:
1058 parsePPElse();
1059 break;
1060 case tok::pp_endif:
1061 parsePPEndIf();
1062 break;
1063 case tok::pp_pragma:
1064 parsePPPragma();
1065 break;
1066 case tok::pp_error:
1067 case tok::pp_warning:
1068 nextToken();
1069 if (!eof() && Style.isCpp())
1070 FormatTok->setFinalizedType(TT_AfterPPDirective);
1071 [[fallthrough]];
1072 default:
1073 parsePPUnknown();
1074 break;
1075 }
1076}
1077
1078void UnwrappedLineParser::conditionalCompilationCondition(bool Unreachable) {
1079 size_t Line = CurrentLines->size();
1080 if (CurrentLines == &PreprocessorDirectives)
1081 Line += Lines.size();
1082
1083 if (Unreachable ||
1084 (!PP.Stack.empty() && PP.Stack.back().Kind == PP_Unreachable)) {
1085 PP.Stack.push_back({PP_Unreachable, Line});
1086 } else {
1087 PP.Stack.push_back({PP_Conditional, Line});
1088 }
1089}
1090
1091void UnwrappedLineParser::conditionalCompilationStart(bool Unreachable) {
1092 ++PP.BranchLevel;
1093 assert(PP.BranchLevel >= 0 &&
1094 PP.BranchLevel <= (int)PP.LevelBranchIndex.size());
1095 if (PP.BranchLevel == (int)PP.LevelBranchIndex.size()) {
1096 PP.LevelBranchIndex.push_back(0);
1097 PP.LevelBranchCount.push_back(0);
1098 }
1099 PP.ChainBranchIndex.push(Unreachable ? -1 : 0);
1100 bool Skip = PP.LevelBranchIndex[PP.BranchLevel] > 0;
1101 conditionalCompilationCondition(Unreachable || Skip);
1102}
1103
1104void UnwrappedLineParser::conditionalCompilationAlternative() {
1105 if (!PP.Stack.empty())
1106 PP.Stack.pop_back();
1107 assert(PP.BranchLevel < (int)PP.LevelBranchIndex.size());
1108 if (!PP.ChainBranchIndex.empty())
1109 ++PP.ChainBranchIndex.top();
1110 conditionalCompilationCondition(
1111 PP.BranchLevel >= 0 && !PP.ChainBranchIndex.empty() &&
1112 PP.LevelBranchIndex[PP.BranchLevel] != PP.ChainBranchIndex.top());
1113}
1114
1115void UnwrappedLineParser::conditionalCompilationEnd() {
1116 assert(PP.BranchLevel < (int)PP.LevelBranchIndex.size());
1117 if (PP.BranchLevel >= 0 && !PP.ChainBranchIndex.empty()) {
1118 if (PP.ChainBranchIndex.top() + 1 > PP.LevelBranchCount[PP.BranchLevel])
1119 PP.LevelBranchCount[PP.BranchLevel] = PP.ChainBranchIndex.top() + 1;
1120 }
1121 // Guard against #endif's without #if.
1122 if (PP.BranchLevel > -1)
1123 --PP.BranchLevel;
1124 if (!PP.ChainBranchIndex.empty())
1125 PP.ChainBranchIndex.pop();
1126 if (!PP.Stack.empty())
1127 PP.Stack.pop_back();
1128}
1129
1130void UnwrappedLineParser::parsePPIf(bool IfDef) {
1131 bool IfNDef = FormatTok->is(tok::pp_ifndef);
1132 nextToken();
1133 bool Unreachable = false;
1134 if (!IfDef && (FormatTok->is(tok::kw_false) || FormatTok->TokenText == "0"))
1135 Unreachable = true;
1136 if (IfDef && !IfNDef && FormatTok->TokenText == "SWIG")
1137 Unreachable = true;
1138 conditionalCompilationStart(Unreachable);
1139 FormatToken *IfCondition = FormatTok;
1140 // If there's a #ifndef on the first line, and the only lines before it are
1141 // comments, it could be an include guard.
1142 bool MaybeIncludeGuard = IfNDef;
1143 if (PP.IncludeGuard == IG_Inited && MaybeIncludeGuard) {
1144 for (auto &Line : Lines) {
1145 if (Line.Tokens.front().Tok->isNot(tok::comment)) {
1146 MaybeIncludeGuard = false;
1147 PP.IncludeGuard = IG_Rejected;
1148 break;
1149 }
1150 }
1151 }
1152 --PP.BranchLevel;
1153 parsePPUnknown();
1154 ++PP.BranchLevel;
1155 if (PP.IncludeGuard == IG_Inited && MaybeIncludeGuard) {
1156 PP.IncludeGuard = IG_IfNdefed;
1157 PP.IncludeGuardToken = IfCondition;
1158 }
1159}
1160
1161void UnwrappedLineParser::parsePPElse() {
1162 // If a potential include guard has an #else, it's not an include guard.
1163 if (PP.IncludeGuard == IG_Defined && PP.BranchLevel == 0)
1164 PP.IncludeGuard = IG_Rejected;
1165 // Don't crash when there is an #else without an #if.
1166 assert(PP.BranchLevel >= -1);
1167 if (PP.BranchLevel == -1)
1168 conditionalCompilationStart(/*Unreachable=*/true);
1169 conditionalCompilationAlternative();
1170 --PP.BranchLevel;
1171 parsePPUnknown();
1172 ++PP.BranchLevel;
1173}
1174
1175void UnwrappedLineParser::parsePPEndIf() {
1176 conditionalCompilationEnd();
1177 parsePPUnknown();
1178}
1179
1180void UnwrappedLineParser::parsePPDefine() {
1181 nextToken();
1182
1183 if (!FormatTok->Tok.getIdentifierInfo()) {
1184 PP.IncludeGuard = IG_Rejected;
1185 PP.IncludeGuardToken = nullptr;
1186 parsePPUnknown();
1187 return;
1188 }
1189
1190 bool MaybeIncludeGuard = false;
1191 if (PP.IncludeGuard == IG_IfNdefed &&
1192 PP.IncludeGuardToken->TokenText == FormatTok->TokenText) {
1193 PP.IncludeGuard = IG_Defined;
1194 PP.IncludeGuardToken = nullptr;
1195 for (auto &Line : Lines) {
1196 if (Line.Tokens.front().Tok->isNoneOf(tok::comment, tok::hash)) {
1197 PP.IncludeGuard = IG_Rejected;
1198 break;
1199 }
1200 }
1201 MaybeIncludeGuard = PP.IncludeGuard == IG_Defined;
1202 }
1203
1204 // In the context of a define, even keywords should be treated as normal
1205 // identifiers. Setting the kind to identifier is not enough, because we need
1206 // to treat additional keywords like __except as well, which are already
1207 // identifiers. Setting the identifier info to null interferes with include
1208 // guard processing above, and changes preprocessing nesting.
1209 FormatTok->Tok.setKind(tok::identifier);
1210 FormatTok->Tok.setIdentifierInfo(Keywords.kw_internal_ident_after_define);
1211 nextToken();
1212
1213 // IncludeGuard can't have a non-empty macro definition.
1214 if (MaybeIncludeGuard && !eof())
1215 PP.IncludeGuard = IG_Rejected;
1216
1217 if (FormatTok->is(tok::l_paren) && !FormatTok->hasWhitespaceBefore())
1218 parseParens();
1219 if (Style.IndentPPDirectives != FormatStyle::PPDIS_None)
1220 Line->Level += PP.BranchLevel + 1;
1221 addUnwrappedLine();
1222 ++Line->Level;
1223
1224 Line->PPLevel = PP.BranchLevel + (PP.IncludeGuard == IG_Defined ? 0 : 1);
1225 assert((int)Line->PPLevel >= 0);
1226
1227 if (eof())
1228 return;
1229
1230 Line->InMacroBody = true;
1231
1232 if (!Style.SkipMacroDefinitionBody) {
1233 // Errors during a preprocessor directive can only affect the layout of the
1234 // preprocessor directive, and thus we ignore them. An alternative approach
1235 // would be to use the same approach we use on the file level (no
1236 // re-indentation if there was a structural error) within the macro
1237 // definition.
1238 parseFile();
1239 return;
1240 }
1241
1242 for (auto *Comment : CommentsBeforeNextToken)
1243 Comment->Finalized = true;
1244
1245 do {
1246 FormatTok->Finalized = true;
1247 FormatTok = Tokens->getNextToken();
1248 } while (!eof());
1249
1250 addUnwrappedLine();
1251}
1252
1253void UnwrappedLineParser::parsePPPragma() {
1254 Line->InPragmaDirective = true;
1255 parsePPUnknown();
1256}
1257
1258void UnwrappedLineParser::parsePPUnknown() {
1259 while (!eof())
1260 nextToken();
1261 if (Style.IndentPPDirectives != FormatStyle::PPDIS_None)
1262 Line->Level += PP.BranchLevel + 1;
1263 addUnwrappedLine();
1264}
1265
1266// Here we exclude certain tokens that are not usually the first token in an
1267// unwrapped line. This is used in attempt to distinguish macro calls without
1268// trailing semicolons from other constructs split to several lines.
1270 // Semicolon can be a null-statement, l_square can be a start of a macro or
1271 // a C++11 attribute, but this doesn't seem to be common.
1272 return Tok.isNoneOf(tok::semi, tok::l_brace,
1273 // Tokens that can only be used as binary operators and a
1274 // part of overloaded operator names.
1275 tok::period, tok::periodstar, tok::arrow, tok::arrowstar,
1276 tok::less, tok::greater, tok::slash, tok::percent,
1277 tok::lessless, tok::greatergreater, tok::equal,
1278 tok::plusequal, tok::minusequal, tok::starequal,
1279 tok::slashequal, tok::percentequal, tok::ampequal,
1280 tok::pipeequal, tok::caretequal, tok::greatergreaterequal,
1281 tok::lesslessequal,
1282 // Colon is used in labels, base class lists, initializer
1283 // lists, range-based for loops, ternary operator, but
1284 // should never be the first token in an unwrapped line.
1285 tok::colon,
1286 // 'noexcept' is a trailing annotation.
1287 tok::kw_noexcept);
1288}
1289
1290static bool mustBeJSIdent(const AdditionalKeywords &Keywords,
1291 const FormatToken *FormatTok) {
1292 // FIXME: This returns true for C/C++ keywords like 'struct'.
1293 return FormatTok->is(tok::identifier) &&
1294 (!FormatTok->Tok.getIdentifierInfo() ||
1295 FormatTok->isNoneOf(
1296 Keywords.kw_in, Keywords.kw_of, Keywords.kw_as, Keywords.kw_async,
1297 Keywords.kw_await, Keywords.kw_yield, Keywords.kw_finally,
1298 Keywords.kw_function, Keywords.kw_import, Keywords.kw_is,
1299 Keywords.kw_let, Keywords.kw_var, tok::kw_const,
1300 Keywords.kw_abstract, Keywords.kw_extends, Keywords.kw_implements,
1301 Keywords.kw_instanceof, Keywords.kw_interface,
1302 Keywords.kw_override, Keywords.kw_throws, Keywords.kw_from));
1303}
1304
1305static bool mustBeJSIdentOrValue(const AdditionalKeywords &Keywords,
1306 const FormatToken *FormatTok) {
1307 return FormatTok->Tok.isLiteral() ||
1308 FormatTok->isOneOf(tok::kw_true, tok::kw_false) ||
1309 mustBeJSIdent(Keywords, FormatTok);
1310}
1311
1312// isJSDeclOrStmt returns true if |FormatTok| starts a declaration or statement
1313// when encountered after a value (see mustBeJSIdentOrValue).
1314static bool isJSDeclOrStmt(const AdditionalKeywords &Keywords,
1315 const FormatToken *FormatTok) {
1316 return FormatTok->isOneOf(
1317 tok::kw_return, Keywords.kw_yield,
1318 // conditionals
1319 tok::kw_if, tok::kw_else,
1320 // loops
1321 tok::kw_for, tok::kw_while, tok::kw_do, tok::kw_continue, tok::kw_break,
1322 // switch/case
1323 tok::kw_switch, tok::kw_case,
1324 // exceptions
1325 tok::kw_throw, tok::kw_try, tok::kw_catch, Keywords.kw_finally,
1326 // declaration
1327 tok::kw_const, tok::kw_class, Keywords.kw_var, Keywords.kw_let,
1328 Keywords.kw_async, Keywords.kw_function,
1329 // import/export
1330 Keywords.kw_import, tok::kw_export);
1331}
1332
1333// Checks whether a token is a type in K&R C (aka C78).
1334static bool isC78Type(const FormatToken &Tok) {
1335 return Tok.isOneOf(tok::kw_char, tok::kw_short, tok::kw_int, tok::kw_long,
1336 tok::kw_unsigned, tok::kw_float, tok::kw_double,
1337 tok::identifier);
1338}
1339
1340// This function checks whether a token starts the first parameter declaration
1341// in a K&R C (aka C78) function definition, e.g.:
1342// int f(a, b)
1343// short a, b;
1344// {
1345// return a + b;
1346// }
1348 const FormatToken *FuncName) {
1349 assert(Tok);
1350 assert(Next);
1351 assert(FuncName);
1352
1353 if (FuncName->isNot(tok::identifier))
1354 return false;
1355
1356 const FormatToken *Prev = FuncName->Previous;
1357 if (!Prev || (Prev->isNot(tok::star) && !isC78Type(*Prev)))
1358 return false;
1359
1360 if (!isC78Type(*Tok) &&
1361 Tok->isNoneOf(tok::kw_register, tok::kw_struct, tok::kw_union)) {
1362 return false;
1363 }
1364
1365 if (Next->isNot(tok::star) && !Next->Tok.getIdentifierInfo())
1366 return false;
1367
1368 Tok = Tok->Previous;
1369 if (!Tok || Tok->isNot(tok::r_paren))
1370 return false;
1371
1372 Tok = Tok->Previous;
1373 if (!Tok || Tok->isNot(tok::identifier))
1374 return false;
1375
1376 return Tok->Previous && Tok->Previous->isOneOf(tok::l_paren, tok::comma);
1377}
1378
1379bool UnwrappedLineParser::parseModuleDecl() {
1380 assert(IsCpp);
1381 assert(FormatTok->is(Keywords.kw_module));
1382
1383 if (Style.Language == FormatStyle::LK_C ||
1384 Style.Standard < FormatStyle::LS_Cpp20) {
1385 return false;
1386 }
1387
1388 nextToken();
1389 if (FormatTok->isNot(tok::identifier))
1390 return false;
1391
1392 for (nextToken(); FormatTok->isNoneOf(tok::semi, tok::eof); nextToken())
1393 if (FormatTok->is(tok::colon))
1394 FormatTok->setFinalizedType(TT_ModulePartitionColon);
1395
1396 nextToken();
1397 Line->IsModuleOrImportDecl = true;
1398 addUnwrappedLine();
1399 return true;
1400}
1401
1402bool UnwrappedLineParser::parseImportDecl() {
1403 assert(IsCpp);
1404 assert(FormatTok->is(Keywords.kw_import) && "'import' expected");
1405
1406 if (Style.Language == FormatStyle::LK_C ||
1407 Style.Standard < FormatStyle::LS_Cpp20) {
1408 return false;
1409 }
1410
1411 nextToken();
1412 if (FormatTok->is(tok::colon)) {
1413 FormatTok->setFinalizedType(TT_ModulePartitionColon);
1414 nextToken();
1415 }
1416 if (FormatTok->isNoneOf(tok::identifier, tok::less, tok::string_literal))
1417 return false;
1418
1419 for (; FormatTok->isNoneOf(tok::semi, tok::eof); nextToken()) {
1420 // Handle import <foo/bar.h> as we would an include statement.
1421 if (FormatTok->is(tok::less)) {
1422 for (nextToken(); FormatTok->isNoneOf(tok::greater, tok::semi, tok::eof);
1423 nextToken()) {
1424 // Mark tokens as implicit string literals, so that import <A/Foo> will
1425 // neither be broken nor have a space added.
1426 FormatTok->setFinalizedType(TT_ImplicitStringLiteral);
1427 }
1428 }
1429 }
1430
1431 nextToken();
1432 Line->IsModuleOrImportDecl = true;
1433 addUnwrappedLine();
1434 return true;
1435}
1436
1437// readTokenWithJavaScriptASI reads the next token and terminates the current
1438// line if JavaScript Automatic Semicolon Insertion must
1439// happen between the current token and the next token.
1440//
1441// This method is conservative - it cannot cover all edge cases of JavaScript,
1442// but only aims to correctly handle certain well known cases. It *must not*
1443// return true in speculative cases.
1444void UnwrappedLineParser::readTokenWithJavaScriptASI() {
1445 FormatToken *Previous = FormatTok;
1446 readToken();
1447 FormatToken *Next = FormatTok;
1448
1449 bool IsOnSameLine =
1450 CommentsBeforeNextToken.empty()
1451 ? Next->NewlinesBefore == 0
1452 : CommentsBeforeNextToken.front()->NewlinesBefore == 0;
1453 if (IsOnSameLine)
1454 return;
1455
1456 bool PreviousMustBeValue = mustBeJSIdentOrValue(Keywords, Previous);
1457 bool PreviousStartsTemplateExpr =
1458 Previous->is(TT_TemplateString) && Previous->TokenText.ends_with("${");
1459 if (PreviousMustBeValue || Previous->is(tok::r_paren)) {
1460 // If the line contains an '@' sign, the previous token might be an
1461 // annotation, which can precede another identifier/value.
1462 bool HasAt = llvm::any_of(Line->Tokens, [](UnwrappedLineNode &LineNode) {
1463 return LineNode.Tok->is(tok::at);
1464 });
1465 if (HasAt)
1466 return;
1467 }
1468 if (Next->is(tok::exclaim) && PreviousMustBeValue)
1469 return addUnwrappedLine();
1470 bool NextMustBeValue = mustBeJSIdentOrValue(Keywords, Next);
1471 bool NextEndsTemplateExpr =
1472 Next->is(TT_TemplateString) && Next->TokenText.starts_with("}");
1473 if (NextMustBeValue && !NextEndsTemplateExpr && !PreviousStartsTemplateExpr &&
1474 (PreviousMustBeValue ||
1475 Previous->isOneOf(tok::r_square, tok::r_paren, tok::plusplus,
1476 tok::minusminus))) {
1477 return addUnwrappedLine();
1478 }
1479 if ((PreviousMustBeValue || Previous->is(tok::r_paren)) &&
1480 isJSDeclOrStmt(Keywords, Next)) {
1481 return addUnwrappedLine();
1482 }
1483}
1484
1485void UnwrappedLineParser::parseStructuralElement(
1486 const FormatToken *OpeningBrace, IfStmtKind *IfKind,
1487 FormatToken **IfLeftBrace, bool *HasDoWhile, bool *HasLabel) {
1488 if (Style.isTableGen() && FormatTok->is(tok::pp_include)) {
1489 nextToken();
1490 if (FormatTok->is(tok::string_literal))
1491 nextToken();
1492 addUnwrappedLine();
1493 return;
1494 }
1495
1496 if (IsCpp) {
1497 while (FormatTok->is(tok::l_square) && handleCppAttributes()) {
1498 }
1499 } else if (Style.isVerilog()) {
1500 // Skip attributes.
1501 while (FormatTok->is(tok::l_paren) &&
1502 Tokens->peekNextToken()->is(tok::star)) {
1503 parseParens();
1504 }
1505 skipVerilogQualifiers();
1506 // Skip things that can exist before keywords like 'if' and 'case'.
1507 if (FormatTok->isOneOf(Keywords.kw_priority, Keywords.kw_unique,
1508 Keywords.kw_unique0)) {
1509 nextToken();
1510 }
1511
1512 if (Keywords.isVerilogStructuredProcedure(*FormatTok)) {
1513 parseForOrWhileLoop(/*HasParens=*/false);
1514 return;
1515 }
1516 if (FormatTok->isOneOf(Keywords.kw_foreach, Keywords.kw_repeat)) {
1517 parseForOrWhileLoop();
1518 return;
1519 }
1520 if (FormatTok->isOneOf(tok::kw_restrict, Keywords.kw_assert,
1521 Keywords.kw_assume, Keywords.kw_cover)) {
1522 parseIfThenElse(IfKind, /*KeepBraces=*/false, /*IsVerilogAssert=*/true);
1523 return;
1524 }
1525 }
1526
1527 // Tokens that only make sense at the beginning of a line.
1528 if (FormatTok->isAccessSpecifierKeyword()) {
1529 if (Style.isJava() || Style.isJavaScript() || Style.isCSharp())
1530 nextToken();
1531 else
1532 parseAccessSpecifier();
1533 return;
1534 }
1535 switch (FormatTok->Tok.getKind()) {
1536 case tok::kw_asm: {
1537 // Track whether to skip formatting inline asm by finalizing the tokens
1538 // in the block. Formatting is skipped inside of braces by default.
1539 // A style option could be added to also skip formatting inside parens.
1540 bool DoNotFormat = false;
1541 tok::TokenKind OpenType;
1542 tok::TokenKind CloseType;
1543 nextToken();
1544 while (FormatTok &&
1545 FormatTok->isOneOf(tok::kw_volatile, tok::kw_inline, tok::kw_goto)) {
1546 nextToken();
1547 }
1548 if (!FormatTok)
1549 break;
1550 if (FormatTok->is(tok::l_brace)) {
1551 FormatTok->setFinalizedType(TT_InlineASMBrace);
1552 OpenType = tok::l_brace;
1553 CloseType = tok::r_brace;
1554 DoNotFormat = true;
1555 } else if (FormatTok->is(tok::l_paren)) {
1556 OpenType = tok::l_paren;
1557 CloseType = tok::r_paren;
1558 FormatTok->setFinalizedType(TT_InlineASMParen);
1559 } else {
1560 break;
1561 }
1562 if (DoNotFormat) {
1563 FormatToken *OpenTok = FormatTok;
1564 int NestLevel = 0;
1565 nextToken();
1566 while (FormatTok && !eof()) {
1567 if (FormatTok->is(OpenType)) {
1568 ++NestLevel;
1569 } else if (FormatTok->is(CloseType)) {
1570 --NestLevel;
1571 if (NestLevel < 1) {
1572 FormatTok->setFinalizedType(OpenTok->getType());
1573 nextToken();
1574 addUnwrappedLine();
1575 break;
1576 }
1577 }
1578 FormatTok->Finalized = true;
1579 nextToken();
1580 }
1581 }
1582 break;
1583 }
1584 case tok::kw_namespace:
1585 parseNamespace();
1586 return;
1587 case tok::kw_if: {
1588 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1589 // field/method declaration.
1590 break;
1591 }
1592 FormatToken *Tok = parseIfThenElse(IfKind);
1593 if (IfLeftBrace)
1594 *IfLeftBrace = Tok;
1595 return;
1596 }
1597 case tok::kw_for:
1598 case tok::kw_while:
1599 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1600 // field/method declaration.
1601 break;
1602 }
1603 parseForOrWhileLoop();
1604 return;
1605 case tok::kw_do:
1606 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1607 // field/method declaration.
1608 break;
1609 }
1610 parseDoWhile();
1611 if (HasDoWhile)
1612 *HasDoWhile = true;
1613 return;
1614 case tok::kw_switch:
1615 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1616 // 'switch: string' field declaration.
1617 break;
1618 }
1619 parseSwitch(/*IsExpr=*/false);
1620 return;
1621 case tok::kw_default: {
1622 // In Verilog default along with other labels are handled in the next loop.
1623 if (Style.isVerilog())
1624 break;
1625 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1626 // 'default: string' field declaration.
1627 break;
1628 }
1629 auto *Default = FormatTok;
1630 nextToken();
1631 if (FormatTok->is(tok::colon)) {
1632 FormatTok->setFinalizedType(TT_CaseLabelColon);
1633 parseLabel();
1634 return;
1635 }
1636 if (FormatTok->is(tok::arrow)) {
1637 FormatTok->setFinalizedType(TT_CaseLabelArrow);
1638 Default->setFinalizedType(TT_SwitchExpressionLabel);
1639 parseLabel();
1640 return;
1641 }
1642 // e.g. "default void f() {}" in a Java interface.
1643 break;
1644 }
1645 case tok::kw_case:
1646 // Proto: there are no switch/case statements.
1647 if (Style.Language == FormatStyle::LK_Proto) {
1648 nextToken();
1649 return;
1650 }
1651 if (Style.isVerilog()) {
1652 parseBlock();
1653 addUnwrappedLine();
1654 return;
1655 }
1656 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1657 // 'case: string' field declaration.
1658 nextToken();
1659 break;
1660 }
1661 parseCaseLabel();
1662 return;
1663 case tok::kw_goto:
1664 nextToken();
1665 if (FormatTok->is(tok::kw_case))
1666 nextToken();
1667 break;
1668 case tok::kw_try:
1669 case tok::kw___try:
1670 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1671 // field/method declaration.
1672 break;
1673 }
1674 parseTryCatch();
1675 return;
1676 case tok::kw_extern:
1677 if (Style.isVerilog()) {
1678 // In Verilog an extern module declaration looks like a start of module.
1679 // But there is no body and endmodule. So we handle it separately.
1680 parseVerilogExtern();
1681 return;
1682 }
1683 nextToken();
1684 if (FormatTok->is(tok::string_literal)) {
1685 nextToken();
1686 if (FormatTok->is(tok::l_brace)) {
1687 if (Style.BraceWrapping.AfterExternBlock)
1688 addUnwrappedLine();
1689 // Either we indent or for backwards compatibility we follow the
1690 // AfterExternBlock style.
1691 unsigned AddLevels =
1692 (Style.IndentExternBlock == FormatStyle::IEBS_Indent) ||
1693 (Style.BraceWrapping.AfterExternBlock &&
1694 Style.IndentExternBlock ==
1696 ? 1u
1697 : 0u;
1698 parseBlock(/*MustBeDeclaration=*/true, AddLevels);
1699 addUnwrappedLine();
1700 return;
1701 }
1702 }
1703 break;
1704 case tok::kw_export:
1705 if (IsCpp) {
1706 nextToken();
1707 if (FormatTok->is(tok::kw_namespace)) {
1708 parseNamespace();
1709 return;
1710 }
1711 if (FormatTok->is(tok::l_brace)) {
1712 parseCppExportBlock();
1713 return;
1714 }
1715 if (FormatTok->is(Keywords.kw_module) && parseModuleDecl())
1716 return;
1717 if (FormatTok->is(Keywords.kw_import) && parseImportDecl())
1718 return;
1719 break;
1720 }
1721 if (Style.isJavaScript()) {
1722 parseJavaScriptEs6ImportExport();
1723 return;
1724 }
1725 if (Style.isVerilog()) {
1726 parseVerilogExtern();
1727 return;
1728 }
1729 break;
1730 case tok::kw_inline:
1731 nextToken();
1732 if (FormatTok->is(tok::kw_namespace)) {
1733 parseNamespace();
1734 return;
1735 }
1736 break;
1737 case tok::identifier:
1738 if (FormatTok->is(TT_ForEachMacro)) {
1739 parseForOrWhileLoop();
1740 return;
1741 }
1742 if (FormatTok->is(TT_MacroBlockBegin)) {
1743 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
1744 /*MunchSemi=*/false);
1745 return;
1746 }
1747 if (FormatTok->is(Keywords.kw_import)) {
1748 if (IsCpp && parseImportDecl())
1749 return;
1750 if (Style.isJavaScript()) {
1751 parseJavaScriptEs6ImportExport();
1752 return;
1753 }
1754 if (Style.Language == FormatStyle::LK_Proto) {
1755 nextToken();
1756 if (FormatTok->is(tok::kw_public))
1757 nextToken();
1758 if (FormatTok->isNot(tok::string_literal))
1759 return;
1760 nextToken();
1761 if (FormatTok->is(tok::semi))
1762 nextToken();
1763 addUnwrappedLine();
1764 return;
1765 }
1766 if (Style.isVerilog()) {
1767 parseVerilogExtern();
1768 return;
1769 }
1770 }
1771 if (IsCpp) {
1772 if (FormatTok->is(Keywords.kw_module) && parseModuleDecl())
1773 return;
1774 if (FormatTok->isOneOf(Keywords.kw_signals, Keywords.kw_qsignals,
1775 Keywords.kw_slots, Keywords.kw_qslots)) {
1776 nextToken();
1777 if (FormatTok->is(tok::colon)) {
1778 nextToken();
1779 addUnwrappedLine();
1780 return;
1781 }
1782 }
1783 if (FormatTok->is(TT_StatementMacro)) {
1784 parseStatementMacro();
1785 return;
1786 }
1787 if (FormatTok->is(TT_NamespaceMacro)) {
1788 parseNamespace();
1789 return;
1790 }
1791 }
1792 // In Verilog labels can be any expression, so we don't do them here.
1793 // JS doesn't have macros, and within classes colons indicate fields, not
1794 // labels.
1795 // TableGen doesn't have labels.
1796 if (!Style.isJavaScript() && !Style.isVerilog() && !Style.isTableGen() &&
1797 Tokens->peekNextToken()->is(tok::colon) && !Line->MustBeDeclaration) {
1798 nextToken();
1799 if (!Line->InMacroBody || CurrentLines->size() > 1)
1800 Line->Tokens.begin()->Tok->MustBreakBefore = true;
1801 FormatTok->setFinalizedType(TT_GotoLabelColon);
1802 parseLabel(/*IsGotoLabel=*/true);
1803 if (HasLabel)
1804 *HasLabel = true;
1805 return;
1806 }
1807 if (Style.isJava() && FormatTok->is(Keywords.kw_record)) {
1808 parseRecord(/*ParseAsExpr=*/false, /*IsJavaRecord=*/true);
1809 addUnwrappedLine();
1810 return;
1811 }
1812 // In all other cases, parse the declaration.
1813 break;
1814 default:
1815 break;
1816 }
1817
1818 bool SeenEqual = false;
1819 for (const bool InRequiresExpression =
1820 OpeningBrace && OpeningBrace->isOneOf(TT_RequiresExpressionLBrace,
1821 TT_CompoundRequirementLBrace);
1822 !eof();) {
1823 const FormatToken *Previous = FormatTok->Previous;
1824 switch (FormatTok->Tok.getKind()) {
1825 case tok::at:
1826 nextToken();
1827 if (FormatTok->is(tok::l_brace)) {
1828 nextToken();
1829 parseBracedList();
1830 break;
1831 }
1832 if (Style.isJava() && FormatTok->is(Keywords.kw_interface)) {
1833 nextToken();
1834 break;
1835 }
1836 switch (bool IsAutoRelease = false; FormatTok->Tok.getObjCKeywordID()) {
1837 case tok::objc_public:
1838 case tok::objc_protected:
1839 case tok::objc_package:
1840 case tok::objc_private:
1841 return parseAccessSpecifier();
1842 case tok::objc_interface:
1843 case tok::objc_implementation:
1844 return parseObjCInterfaceOrImplementation();
1845 case tok::objc_protocol:
1846 if (parseObjCProtocol())
1847 return;
1848 break;
1849 case tok::objc_end:
1850 return; // Handled by the caller.
1851 case tok::objc_optional:
1852 case tok::objc_required:
1853 nextToken();
1854 addUnwrappedLine();
1855 return;
1856 case tok::objc_autoreleasepool:
1857 IsAutoRelease = true;
1858 [[fallthrough]];
1859 case tok::objc_synchronized:
1860 nextToken();
1861 if (!IsAutoRelease && FormatTok->is(tok::l_paren)) {
1862 // Skip synchronization object
1863 parseParens();
1864 }
1865 if (FormatTok->is(tok::l_brace)) {
1866 if (Style.BraceWrapping.AfterControlStatement ==
1868 addUnwrappedLine();
1869 }
1870 parseBlock();
1871 }
1872 addUnwrappedLine();
1873 return;
1874 case tok::objc_try:
1875 // This branch isn't strictly necessary (the kw_try case below would
1876 // do this too after the tok::at is parsed above). But be explicit.
1877 parseTryCatch();
1878 return;
1879 default:
1880 break;
1881 }
1882 break;
1883 case tok::kw_requires: {
1884 if (IsCpp) {
1885 bool ParsedClause = parseRequires(SeenEqual);
1886 if (ParsedClause)
1887 return;
1888 } else {
1889 nextToken();
1890 }
1891 break;
1892 }
1893 case tok::kw_enum:
1894 // Ignore if this is part of "template <enum ..." or "... -> enum" or
1895 // "template <..., enum ...>".
1896 if (Previous && Previous->isOneOf(tok::less, tok::arrow, tok::comma)) {
1897 nextToken();
1898 break;
1899 }
1900
1901 // parseEnum falls through and does not yet add an unwrapped line as an
1902 // enum definition can start a structural element.
1903 if (!parseEnum())
1904 break;
1905 // This only applies to C++ and Verilog.
1906 if (!IsCpp && !Style.isVerilog()) {
1907 addUnwrappedLine();
1908 return;
1909 }
1910 break;
1911 case tok::kw_typedef:
1912 nextToken();
1913 if (FormatTok->isOneOf(Keywords.kw_NS_ENUM, Keywords.kw_NS_OPTIONS,
1914 Keywords.kw_CF_ENUM, Keywords.kw_CF_OPTIONS,
1915 Keywords.kw_CF_CLOSED_ENUM,
1916 Keywords.kw_NS_CLOSED_ENUM)) {
1917 parseEnum();
1918 }
1919 break;
1920 case tok::kw_class:
1921 if (Style.isVerilog()) {
1922 parseBlock();
1923 addUnwrappedLine();
1924 return;
1925 }
1926 if (Style.isTableGen()) {
1927 // Do nothing special. In this case the l_brace becomes FunctionLBrace.
1928 // This is same as def and so on.
1929 nextToken();
1930 break;
1931 }
1932 [[fallthrough]];
1933 case tok::kw_struct:
1934 case tok::kw_union:
1935 if (parseStructLike())
1936 return;
1937 break;
1938 case tok::kw_decltype:
1939 nextToken();
1940 if (FormatTok->is(tok::l_paren)) {
1941 parseParens();
1942 if (FormatTok->Previous &&
1943 FormatTok->Previous->endsSequence(tok::r_paren, tok::kw_auto,
1944 tok::l_paren)) {
1945 Line->SeenDecltypeAuto = true;
1946 }
1947 }
1948 break;
1949 case tok::period:
1950 nextToken();
1951 // In Java, classes have an implicit static member "class".
1952 if (Style.isJava() && FormatTok && FormatTok->is(tok::kw_class))
1953 nextToken();
1954 if (Style.isJavaScript() && FormatTok &&
1955 FormatTok->Tok.getIdentifierInfo()) {
1956 // JavaScript only has pseudo keywords, all keywords are allowed to
1957 // appear in "IdentifierName" positions. See http://es5.github.io/#x7.6
1958 nextToken();
1959 }
1960 break;
1961 case tok::semi:
1962 nextToken();
1963 addUnwrappedLine();
1964 return;
1965 case tok::r_brace:
1966 addUnwrappedLine();
1967 return;
1968 case tok::string_literal:
1969 if (Style.isVerilog() && FormatTok->is(TT_VerilogProtected)) {
1970 FormatTok->Finalized = true;
1971 nextToken();
1972 addUnwrappedLine();
1973 return;
1974 }
1975 nextToken();
1976 break;
1977 case tok::l_paren: {
1978 parseParens();
1979 // Break the unwrapped line if a K&R C function definition has a parameter
1980 // declaration.
1981 if (OpeningBrace || !IsCpp || !Previous || eof())
1982 break;
1983 if (isC78ParameterDecl(FormatTok,
1984 Tokens->peekNextToken(/*SkipComment=*/true),
1985 Previous)) {
1986 addUnwrappedLine();
1987 return;
1988 }
1989 break;
1990 }
1991 case tok::kw_operator:
1992 nextToken();
1993 if (FormatTok->isBinaryOperator())
1994 nextToken();
1995 break;
1996 case tok::caret: {
1997 const auto *Prev = FormatTok->getPreviousNonComment();
1998 nextToken();
1999 if (Prev && Prev->is(tok::identifier))
2000 break;
2001 // Block return type.
2002 if (FormatTok->Tok.isAnyIdentifier() || FormatTok->isTypeName(LangOpts)) {
2003 nextToken();
2004 // Return types: ObjC generics and protocol qualifiers are ok too.
2005 if (FormatTok->is(tok::less)) {
2006 nextToken();
2007 parseBracedList(/*IsAngleBracket=*/true);
2008 }
2009 // Return types: pointers are ok too.
2010 while (FormatTok->is(tok::star))
2011 nextToken();
2012 }
2013 // Block argument list.
2014 if (FormatTok->is(tok::l_paren))
2015 parseParens();
2016 // Block body.
2017 if (FormatTok->is(tok::l_brace))
2018 parseChildBlock();
2019 break;
2020 }
2021 case tok::l_brace:
2022 if (InRequiresExpression)
2023 FormatTok->setFinalizedType(TT_BracedListLBrace);
2024 if (!tryToParsePropertyAccessor() && !tryToParseBracedList()) {
2025 IsDecltypeAutoFunction = Line->SeenDecltypeAuto;
2026 // A block outside of parentheses must be the last part of a
2027 // structural element.
2028 // FIXME: Figure out cases where this is not true, and add projections
2029 // for them (the one we know is missing are lambdas).
2030 if (Style.isJava() &&
2031 Line->Tokens.front().Tok->is(Keywords.kw_synchronized)) {
2032 // If necessary, we could set the type to something different than
2033 // TT_FunctionLBrace.
2034 if (Style.BraceWrapping.AfterControlStatement ==
2036 addUnwrappedLine();
2037 }
2038 } else if (Style.BraceWrapping.AfterFunction) {
2039 addUnwrappedLine();
2040 }
2041 if (!Previous || Previous->isNot(TT_TypeDeclarationParen))
2042 FormatTok->setFinalizedType(TT_FunctionLBrace);
2043 parseBlock();
2044 IsDecltypeAutoFunction = false;
2045 addUnwrappedLine();
2046 return;
2047 }
2048 // Otherwise this was a braced init list, and the structural
2049 // element continues.
2050 break;
2051 case tok::kw_try:
2052 if (Style.isJavaScript() && Line->MustBeDeclaration) {
2053 // field/method declaration.
2054 nextToken();
2055 break;
2056 }
2057 // We arrive here when parsing function-try blocks.
2058 if (Style.BraceWrapping.AfterFunction)
2059 addUnwrappedLine();
2060 parseTryCatch();
2061 return;
2062 case tok::identifier: {
2063 if (Style.isCSharp() && FormatTok->is(Keywords.kw_where) &&
2064 Line->MustBeDeclaration) {
2065 addUnwrappedLine();
2066 parseCSharpGenericTypeConstraint();
2067 break;
2068 }
2069 if (FormatTok->is(TT_MacroBlockEnd)) {
2070 addUnwrappedLine();
2071 return;
2072 }
2073
2074 // Function declarations (as opposed to function expressions) are parsed
2075 // on their own unwrapped line by continuing this loop. Function
2076 // expressions (functions that are not on their own line) must not create
2077 // a new unwrapped line, so they are special cased below.
2078 size_t TokenCount = Line->Tokens.size();
2079 if (Style.isJavaScript() && FormatTok->is(Keywords.kw_function) &&
2080 (TokenCount > 1 ||
2081 (TokenCount == 1 &&
2082 Line->Tokens.front().Tok->isNot(Keywords.kw_async)))) {
2083 tryToParseJSFunction();
2084 break;
2085 }
2086 if ((Style.isJavaScript() || Style.isJava()) &&
2087 FormatTok->is(Keywords.kw_interface)) {
2088 if (Style.isJavaScript()) {
2089 // In JavaScript/TypeScript, "interface" can be used as a standalone
2090 // identifier, e.g. in `var interface = 1;`. If "interface" is
2091 // followed by another identifier, it is very like to be an actual
2092 // interface declaration.
2093 unsigned StoredPosition = Tokens->getPosition();
2094 FormatToken *Next = Tokens->getNextToken();
2095 FormatTok = Tokens->setPosition(StoredPosition);
2096 if (!mustBeJSIdent(Keywords, Next)) {
2097 nextToken();
2098 break;
2099 }
2100 }
2101 parseRecord();
2102 addUnwrappedLine();
2103 return;
2104 }
2105
2106 if (Style.isVerilog()) {
2107 if (FormatTok->is(Keywords.kw_table)) {
2108 parseVerilogTable();
2109 return;
2110 }
2111 if (Keywords.isVerilogBegin(*FormatTok) ||
2112 Keywords.isVerilogHierarchy(*FormatTok)) {
2113 parseBlock();
2114 addUnwrappedLine();
2115 return;
2116 }
2117 }
2118
2119 if (!IsCpp && FormatTok->is(Keywords.kw_interface)) {
2120 if (parseStructLike())
2121 return;
2122 break;
2123 }
2124
2125 if (IsCpp && FormatTok->is(TT_StatementMacro)) {
2126 parseStatementMacro();
2127 return;
2128 }
2129
2130 // See if the following token should start a new unwrapped line.
2131 StringRef Text = FormatTok->TokenText;
2132
2133 FormatToken *PreviousToken = FormatTok;
2134 nextToken();
2135
2136 // JS doesn't have macros, and within classes colons indicate fields, not
2137 // labels.
2138 if (Style.isJavaScript())
2139 break;
2140
2141 auto OneTokenSoFar = [&]() {
2142 auto I = Line->Tokens.begin(), E = Line->Tokens.end();
2143 while (I != E && I->Tok->is(tok::comment))
2144 ++I;
2145 if (Style.isVerilog())
2146 while (I != E && I->Tok->is(tok::hash))
2147 ++I;
2148 return I != E && (++I == E);
2149 };
2150 if (OneTokenSoFar()) {
2151 // Recognize function-like macro usages without trailing semicolon as
2152 // well as free-standing macros like Q_OBJECT.
2153 bool FunctionLike = FormatTok->is(tok::l_paren);
2154 if (FunctionLike)
2155 parseParens();
2156
2157 bool FollowedByNewline =
2158 CommentsBeforeNextToken.empty()
2159 ? FormatTok->NewlinesBefore > 0
2160 : CommentsBeforeNextToken.front()->NewlinesBefore > 0;
2161
2162 if (FollowedByNewline &&
2163 (Text.size() >= 5 ||
2164 (FunctionLike && FormatTok->isNot(tok::l_paren))) &&
2165 tokenCanStartNewLine(*FormatTok) && Text == Text.upper()) {
2166 if (PreviousToken->isNot(TT_UntouchableMacroFunc))
2167 PreviousToken->setFinalizedType(TT_FunctionLikeOrFreestandingMacro);
2168 addUnwrappedLine();
2169 return;
2170 }
2171 }
2172 break;
2173 }
2174 case tok::equal:
2175 if ((Style.isJavaScript() || Style.isCSharp()) &&
2176 FormatTok->is(TT_FatArrow)) {
2177 tryToParseChildBlock();
2178 break;
2179 }
2180
2181 SeenEqual = true;
2182 nextToken();
2183 if (FormatTok->is(tok::l_brace)) {
2184 // C# needs this change to ensure that array initialisers and object
2185 // initialisers are indented the same way. In TypeScript, the brace
2186 // can also be an object type definition.
2187 if (!Style.isJavaScript())
2188 FormatTok->setBlockKind(BK_BracedInit);
2189 // TableGen's defset statement has syntax of the form,
2190 // `defset <type> <name> = { <statement>... }`
2191 if (Style.isTableGen() &&
2192 Line->Tokens.begin()->Tok->is(Keywords.kw_defset)) {
2193 FormatTok->setFinalizedType(TT_FunctionLBrace);
2194 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
2195 /*MunchSemi=*/false);
2196 addUnwrappedLine();
2197 break;
2198 }
2199 nextToken();
2200 parseBracedList();
2201 } else if (Style.Language == FormatStyle::LK_Proto &&
2202 FormatTok->is(tok::less)) {
2203 nextToken();
2204 parseBracedList(/*IsAngleBracket=*/true);
2205 }
2206 break;
2207 case tok::l_square:
2208 parseSquare();
2209 break;
2210 case tok::kw_new:
2211 if (Style.isCSharp() &&
2212 (Tokens->peekNextToken()->isAccessSpecifierKeyword() ||
2213 (Previous && Previous->isAccessSpecifierKeyword()))) {
2214 nextToken();
2215 } else {
2216 parseNew();
2217 }
2218 break;
2219 case tok::kw_switch:
2220 if (Style.isJava())
2221 parseSwitch(/*IsExpr=*/true);
2222 else
2223 nextToken();
2224 break;
2225 case tok::kw_case:
2226 // Proto: there are no switch/case statements.
2227 if (Style.Language == FormatStyle::LK_Proto) {
2228 nextToken();
2229 return;
2230 }
2231 // In Verilog switch is called case.
2232 if (Style.isVerilog()) {
2233 parseBlock();
2234 addUnwrappedLine();
2235 return;
2236 }
2237 if (Style.isJavaScript() && Line->MustBeDeclaration) {
2238 // 'case: string' field declaration.
2239 nextToken();
2240 break;
2241 }
2242 parseCaseLabel();
2243 break;
2244 case tok::kw_default:
2245 nextToken();
2246 if (Style.isVerilog()) {
2247 if (FormatTok->is(tok::colon)) {
2248 // The label will be handled in the next iteration.
2249 break;
2250 }
2251 if (FormatTok->is(Keywords.kw_clocking)) {
2252 // A default clocking block.
2253 parseBlock();
2254 addUnwrappedLine();
2255 return;
2256 }
2257 parseVerilogCaseLabel();
2258 return;
2259 }
2260 break;
2261 case tok::colon:
2262 nextToken();
2263 if (Style.isVerilog()) {
2264 parseVerilogCaseLabel();
2265 return;
2266 }
2267 break;
2268 case tok::greater:
2269 nextToken();
2270 if (FormatTok->is(tok::l_brace))
2271 FormatTok->Previous->setFinalizedType(TT_TemplateCloser);
2272 break;
2273 default:
2274 nextToken();
2275 break;
2276 }
2277 }
2278}
2279
2280bool UnwrappedLineParser::tryToParsePropertyAccessor() {
2281 assert(FormatTok->is(tok::l_brace));
2282 if (!Style.isCSharp())
2283 return false;
2284 // See if it's a property accessor.
2285 if (!FormatTok->Previous || FormatTok->Previous->isNot(tok::identifier))
2286 return false;
2287
2288 // See if we are inside a property accessor.
2289 //
2290 // Record the current tokenPosition so that we can advance and
2291 // reset the current token. `Next` is not set yet so we need
2292 // another way to advance along the token stream.
2293 unsigned int StoredPosition = Tokens->getPosition();
2294 FormatToken *Tok = Tokens->getNextToken();
2295
2296 // A trivial property accessor is of the form:
2297 // { [ACCESS_SPECIFIER] [get]; [ACCESS_SPECIFIER] [set|init] }
2298 // Track these as they do not require line breaks to be introduced.
2299 bool HasSpecialAccessor = false;
2300 bool IsTrivialPropertyAccessor = true;
2301 bool HasAttribute = false;
2302 while (!eof()) {
2303 if (const bool IsAccessorKeyword =
2304 Tok->isOneOf(Keywords.kw_get, Keywords.kw_init, Keywords.kw_set);
2305 IsAccessorKeyword || Tok->isAccessSpecifierKeyword() ||
2306 Tok->isOneOf(tok::l_square, tok::semi, Keywords.kw_internal)) {
2307 if (IsAccessorKeyword)
2308 HasSpecialAccessor = true;
2309 else if (Tok->is(tok::l_square))
2310 HasAttribute = true;
2311 Tok = Tokens->getNextToken();
2312 continue;
2313 }
2314 if (Tok->isNot(tok::r_brace))
2315 IsTrivialPropertyAccessor = false;
2316 break;
2317 }
2318
2319 if (!HasSpecialAccessor || HasAttribute) {
2320 Tokens->setPosition(StoredPosition);
2321 return false;
2322 }
2323
2324 // Try to parse the property accessor:
2325 // https://docs.microsoft.com/en-us/dotnet/csharp/programming-guide/classes-and-structs/properties
2326 Tokens->setPosition(StoredPosition);
2327 if (!IsTrivialPropertyAccessor && Style.BraceWrapping.AfterFunction)
2328 addUnwrappedLine();
2329 nextToken();
2330 do {
2331 switch (FormatTok->Tok.getKind()) {
2332 case tok::r_brace:
2333 nextToken();
2334 if (FormatTok->is(tok::equal)) {
2335 while (!eof() && FormatTok->isNot(tok::semi))
2336 nextToken();
2337 nextToken();
2338 }
2339 addUnwrappedLine();
2340 return true;
2341 case tok::l_brace:
2342 ++Line->Level;
2343 parseBlock(/*MustBeDeclaration=*/true);
2344 addUnwrappedLine();
2345 --Line->Level;
2346 break;
2347 case tok::equal:
2348 if (FormatTok->is(TT_FatArrow)) {
2349 ++Line->Level;
2350 do {
2351 nextToken();
2352 } while (!eof() && FormatTok->isNot(tok::semi));
2353 nextToken();
2354 addUnwrappedLine();
2355 --Line->Level;
2356 break;
2357 }
2358 nextToken();
2359 break;
2360 default:
2361 if (FormatTok->isOneOf(Keywords.kw_get, Keywords.kw_init,
2362 Keywords.kw_set) &&
2363 !IsTrivialPropertyAccessor) {
2364 // Non-trivial get/set needs to be on its own line.
2365 addUnwrappedLine();
2366 }
2367 nextToken();
2368 }
2369 } while (!eof());
2370
2371 // Unreachable for well-formed code (paired '{' and '}').
2372 return true;
2373}
2374
2375bool UnwrappedLineParser::tryToParseLambda() {
2376 assert(FormatTok->is(tok::l_square));
2377 if (!IsCpp) {
2378 nextToken();
2379 return false;
2380 }
2381 FormatToken &LSquare = *FormatTok;
2382 if (!tryToParseLambdaIntroducer())
2383 return false;
2384
2385 FormatToken *Arrow = nullptr;
2386 bool InTemplateParameterList = false;
2387
2388 while (FormatTok->isNot(tok::l_brace)) {
2389 if (FormatTok->isTypeName(LangOpts) || FormatTok->isAttribute()) {
2390 nextToken();
2391 continue;
2392 }
2393 switch (FormatTok->Tok.getKind()) {
2394 case tok::l_brace:
2395 break;
2396 case tok::l_paren:
2397 parseParens(/*AmpAmpTokenType=*/TT_PointerOrReference);
2398 break;
2399 case tok::l_square:
2400 parseSquare();
2401 break;
2402 case tok::less:
2403 assert(FormatTok->Previous);
2404 if (FormatTok->Previous->is(tok::r_square))
2405 InTemplateParameterList = true;
2406 nextToken();
2407 break;
2408 case tok::kw_auto:
2409 case tok::kw_class:
2410 case tok::kw_struct:
2411 case tok::kw_union:
2412 case tok::kw_template:
2413 case tok::kw_typename:
2414 case tok::amp:
2415 case tok::star:
2416 case tok::kw_const:
2417 case tok::kw_constexpr:
2418 case tok::kw_consteval:
2419 case tok::comma:
2420 case tok::greater:
2421 case tok::identifier:
2422 case tok::numeric_constant:
2423 case tok::coloncolon:
2424 case tok::kw_mutable:
2425 case tok::kw_noexcept:
2426 case tok::kw_static:
2427 nextToken();
2428 break;
2429 // Specialization of a template with an integer parameter can contain
2430 // arithmetic, logical, comparison and ternary operators.
2431 //
2432 // FIXME: This also accepts sequences of operators that are not in the scope
2433 // of a template argument list.
2434 //
2435 // In a C++ lambda a template type can only occur after an arrow. We use
2436 // this as an heuristic to distinguish between Objective-C expressions
2437 // followed by an `a->b` expression, such as:
2438 // ([obj func:arg] + a->b)
2439 // Otherwise the code below would parse as a lambda.
2440 case tok::plus:
2441 case tok::minus:
2442 case tok::exclaim:
2443 case tok::tilde:
2444 case tok::slash:
2445 case tok::percent:
2446 case tok::lessless:
2447 case tok::pipe:
2448 case tok::pipepipe:
2449 case tok::ampamp:
2450 case tok::caret:
2451 case tok::equalequal:
2452 case tok::exclaimequal:
2453 case tok::greaterequal:
2454 case tok::lessequal:
2455 case tok::question:
2456 case tok::colon:
2457 case tok::ellipsis:
2458 case tok::kw_true:
2459 case tok::kw_false:
2460 if (Arrow || InTemplateParameterList) {
2461 nextToken();
2462 break;
2463 }
2464 return true;
2465 case tok::arrow:
2466 Arrow = FormatTok;
2467 nextToken();
2468 break;
2469 case tok::kw_requires:
2470 parseRequiresClause();
2471 break;
2472 case tok::equal:
2473 if (!InTemplateParameterList)
2474 return true;
2475 nextToken();
2476 break;
2477 default:
2478 return true;
2479 }
2480 }
2481
2482 FormatTok->setFinalizedType(TT_LambdaLBrace);
2483 LSquare.setFinalizedType(TT_LambdaLSquare);
2484
2485 if (Arrow)
2486 Arrow->setFinalizedType(TT_LambdaArrow);
2487
2488 NestedLambdas.push_back(Line->SeenDecltypeAuto);
2489 parseChildBlock();
2490 assert(!NestedLambdas.empty());
2491 NestedLambdas.pop_back();
2492
2493 return true;
2494}
2495
2496bool UnwrappedLineParser::tryToParseLambdaIntroducer() {
2497 const FormatToken *Previous = FormatTok->Previous;
2498 const FormatToken *LeftSquare = FormatTok;
2499 nextToken();
2500 if (Previous) {
2501 const auto *PrevPrev = Previous->getPreviousNonComment();
2502 if (Previous->is(tok::star) && PrevPrev && PrevPrev->isTypeName(LangOpts))
2503 return false;
2504 if (Previous->closesScope()) {
2505 // Not a potential C-style cast.
2506 if (Previous->isNot(tok::r_paren))
2507 return false;
2508 // Lambdas can be cast to function types only, e.g. `std::function<int()>`
2509 // and `int (*)()`.
2510 if (!PrevPrev || PrevPrev->isNoneOf(tok::greater, tok::r_paren))
2511 return false;
2512 }
2513 if (Previous && Previous->Tok.getIdentifierInfo() &&
2514 Previous->isNoneOf(tok::kw_return, tok::kw_co_await, tok::kw_co_yield,
2515 tok::kw_co_return)) {
2516 return false;
2517 }
2518 }
2519 if (LeftSquare->isCppStructuredBinding(IsCpp))
2520 return false;
2521 if (FormatTok->is(tok::l_square) || tok::isLiteral(FormatTok->Tok.getKind()))
2522 return false;
2523 if (FormatTok->is(tok::r_square)) {
2524 const FormatToken *Next = Tokens->peekNextToken(/*SkipComment=*/true);
2525 if (Next->is(tok::greater))
2526 return false;
2527 }
2528 parseSquare(/*LambdaIntroducer=*/true);
2529 return true;
2530}
2531
2532void UnwrappedLineParser::tryToParseJSFunction() {
2533 assert(FormatTok->is(Keywords.kw_function));
2534 if (FormatTok->is(Keywords.kw_async))
2535 nextToken();
2536 // Consume "function".
2537 nextToken();
2538
2539 // Consume * (generator function). Treat it like C++'s overloaded operators.
2540 if (FormatTok->is(tok::star)) {
2541 FormatTok->setFinalizedType(TT_OverloadedOperator);
2542 nextToken();
2543 }
2544
2545 // Consume function name.
2546 if (FormatTok->is(tok::identifier))
2547 nextToken();
2548
2549 if (FormatTok->isNot(tok::l_paren))
2550 return;
2551
2552 // Parse formal parameter list.
2553 parseParens();
2554
2555 if (FormatTok->is(tok::colon)) {
2556 // Parse a type definition.
2557 nextToken();
2558
2559 // Eat the type declaration. For braced inline object types, balance braces,
2560 // otherwise just parse until finding an l_brace for the function body.
2561 if (FormatTok->is(tok::l_brace))
2562 tryToParseBracedList();
2563 else
2564 while (FormatTok->isNoneOf(tok::l_brace, tok::semi) && !eof())
2565 nextToken();
2566 }
2567
2568 if (FormatTok->is(tok::semi))
2569 return;
2570
2571 parseChildBlock();
2572}
2573
2574bool UnwrappedLineParser::tryToParseBracedList() {
2575 if (FormatTok->is(BK_Unknown))
2576 calculateBraceTypes();
2577 assert(FormatTok->isNot(BK_Unknown));
2578 if (FormatTok->is(BK_Block))
2579 return false;
2580 nextToken();
2581 parseBracedList();
2582 return true;
2583}
2584
2585bool UnwrappedLineParser::tryToParseChildBlock() {
2586 assert(Style.isJavaScript() || Style.isCSharp());
2587 assert(FormatTok->is(TT_FatArrow));
2588 // Fat arrows (=>) have tok::TokenKind tok::equal but TokenType TT_FatArrow.
2589 // They always start an expression or a child block if followed by a curly
2590 // brace.
2591 nextToken();
2592 if (FormatTok->isNot(tok::l_brace))
2593 return false;
2594 parseChildBlock();
2595 return true;
2596}
2597
2598bool UnwrappedLineParser::parseBracedList(bool IsAngleBracket, bool IsEnum) {
2599 assert(!IsAngleBracket || !IsEnum);
2600 bool HasError = false;
2601
2602 // FIXME: Once we have an expression parser in the UnwrappedLineParser,
2603 // replace this by using parseAssignmentExpression() inside.
2604 do {
2605 if (Style.isCSharp() && FormatTok->is(TT_FatArrow) &&
2606 tryToParseChildBlock()) {
2607 continue;
2608 }
2609 if (Style.isJavaScript()) {
2610 if (FormatTok->is(Keywords.kw_function)) {
2611 tryToParseJSFunction();
2612 continue;
2613 }
2614 if (FormatTok->is(tok::l_brace)) {
2615 // Could be a method inside of a braced list `{a() { return 1; }}`.
2616 if (tryToParseBracedList())
2617 continue;
2618 parseChildBlock();
2619 }
2620 }
2621 if (FormatTok->is(IsAngleBracket ? tok::greater : tok::r_brace)) {
2622 if (IsEnum) {
2623 FormatTok->setBlockKind(BK_Block);
2624 if (!Style.AllowShortEnumsOnASingleLine)
2625 addUnwrappedLine();
2626 }
2627 nextToken();
2628 return !HasError;
2629 }
2630 switch (FormatTok->Tok.getKind()) {
2631 case tok::l_square:
2632 if (Style.isCSharp())
2633 parseSquare();
2634 else
2635 tryToParseLambda();
2636 break;
2637 case tok::l_paren:
2638 parseParens();
2639 // JavaScript can just have free standing methods and getters/setters in
2640 // object literals. Detect them by a "{" following ")".
2641 if (Style.isJavaScript()) {
2642 if (FormatTok->is(tok::l_brace))
2643 parseChildBlock();
2644 break;
2645 }
2646 break;
2647 case tok::l_brace:
2648 // Assume there are no blocks inside a braced init list apart
2649 // from the ones we explicitly parse out (like lambdas).
2650 FormatTok->setBlockKind(BK_BracedInit);
2651 if (!IsAngleBracket) {
2652 auto *Prev = FormatTok->Previous;
2653 if (Prev && Prev->is(tok::greater))
2654 Prev->setFinalizedType(TT_TemplateCloser);
2655 }
2656 nextToken();
2657 parseBracedList();
2658 break;
2659 case tok::less:
2660 nextToken();
2661 if (IsAngleBracket)
2662 parseBracedList(/*IsAngleBracket=*/true);
2663 break;
2664 case tok::semi:
2665 // JavaScript (or more precisely TypeScript) can have semicolons in braced
2666 // lists (in so-called TypeMemberLists). Thus, the semicolon cannot be
2667 // used for error recovery if we have otherwise determined that this is
2668 // a braced list.
2669 if (Style.isJavaScript()) {
2670 nextToken();
2671 break;
2672 }
2673 HasError = true;
2674 if (!IsEnum)
2675 return false;
2676 nextToken();
2677 break;
2678 case tok::comma:
2679 nextToken();
2680 if (IsEnum && !Style.AllowShortEnumsOnASingleLine)
2681 addUnwrappedLine();
2682 break;
2683 case tok::kw_requires:
2684 parseRequiresExpression();
2685 break;
2686 default:
2687 nextToken();
2688 break;
2689 }
2690 } while (!eof());
2691 return false;
2692}
2693
2694/// Parses a pair of parentheses (and everything between them).
2695/// \param StarAndAmpTokenType If different than TT_Unknown sets this type for
2696/// all (double) ampersands and stars. This applies for all nested scopes as
2697/// well, this is disabled within a (potential) template argument <>, and thus
2698/// also if we find only a <.
2699///
2700/// Returns whether there is a `=` token between the parentheses.
2701bool UnwrappedLineParser::parseParens(TokenType StarAndAmpTokenType,
2702 bool InMacroCall) {
2703 assert(FormatTok->is(tok::l_paren) && "'(' expected.");
2704 auto *LParen = FormatTok;
2705 auto *Prev = FormatTok->Previous;
2706 bool SeenComma = false;
2707 bool SeenEqual = false;
2708 bool MightBeFoldExpr = false;
2709 unsigned ExcessLess = 0;
2710 nextToken();
2711 const bool MightBeStmtExpr = FormatTok->is(tok::l_brace);
2712 if (!InMacroCall && Prev && Prev->is(TT_FunctionLikeMacro))
2713 InMacroCall = true;
2714 do {
2715 switch (FormatTok->Tok.getKind()) {
2716 case tok::l_paren:
2717 if (parseParens(ExcessLess == 0 ? StarAndAmpTokenType : TT_Unknown,
2718 InMacroCall)) {
2719 SeenEqual = true;
2720 }
2721 if (Style.isJava() && FormatTok->is(tok::l_brace))
2722 parseChildBlock();
2723 break;
2724 case tok::r_paren: {
2725 auto *RParen = FormatTok;
2726 nextToken();
2727 if (Prev) {
2728 auto OptionalParens = [&] {
2729 if (Style.RemoveParentheses == FormatStyle::RPS_Leave ||
2730 MightBeStmtExpr || MightBeFoldExpr || SeenComma || InMacroCall ||
2731 Line->InMacroBody || RParen->getPreviousNonComment() == LParen) {
2732 return false;
2733 }
2734 const bool DoubleParens =
2735 Prev->is(tok::l_paren) && FormatTok->is(tok::r_paren);
2736 if (DoubleParens) {
2737 const auto *PrevPrev = Prev->getPreviousNonComment();
2738 const bool Excluded =
2739 PrevPrev &&
2740 (PrevPrev->isOneOf(tok::kw___attribute, tok::kw_decltype) ||
2741 (SeenEqual &&
2742 (PrevPrev->isOneOf(tok::kw_if, tok::kw_while) ||
2743 PrevPrev->endsSequence(tok::kw_constexpr, tok::kw_if))));
2744 if (!Excluded)
2745 return true;
2746 } else {
2747 const bool CommaSeparated =
2748 Prev->isOneOf(tok::l_paren, tok::comma) &&
2749 FormatTok->isOneOf(tok::comma, tok::r_paren);
2750 if (CommaSeparated &&
2751 // LParen is not preceded by ellipsis, comma.
2752 !Prev->endsSequence(tok::comma, tok::ellipsis) &&
2753 // RParen is not followed by comma, ellipsis.
2754 !(FormatTok->is(tok::comma) &&
2755 Tokens->peekNextToken()->is(tok::ellipsis))) {
2756 return true;
2757 }
2758 const bool ReturnParens =
2759 Style.RemoveParentheses == FormatStyle::RPS_ReturnStatement &&
2760 ((NestedLambdas.empty() && !IsDecltypeAutoFunction) ||
2761 (!NestedLambdas.empty() && !NestedLambdas.back())) &&
2762 Prev->isOneOf(tok::kw_return, tok::kw_co_return) &&
2763 FormatTok->is(tok::semi);
2764 if (ReturnParens)
2765 return true;
2766 }
2767 return false;
2768 };
2769 if (OptionalParens()) {
2770 LParen->Optional = true;
2771 RParen->Optional = true;
2772 } else if (Prev->is(TT_TypenameMacro)) {
2773 LParen->setFinalizedType(TT_TypeDeclarationParen);
2774 RParen->setFinalizedType(TT_TypeDeclarationParen);
2775 } else if (Prev->is(tok::greater) && RParen->Previous == LParen) {
2776 Prev->setFinalizedType(TT_TemplateCloser);
2777 } else if (FormatTok->is(tok::l_brace) && Prev->is(tok::amp) &&
2778 !Prev->Previous) {
2779 FormatTok->setBlockKind(BK_BracedInit);
2780 }
2781 }
2782 return SeenEqual;
2783 }
2784 case tok::r_brace:
2785 // A "}" inside parenthesis is an error if there wasn't a matching "{".
2786 return SeenEqual;
2787 case tok::l_square:
2788 tryToParseLambda();
2789 break;
2790 case tok::l_brace:
2791 if (!tryToParseBracedList())
2792 parseChildBlock();
2793 break;
2794 case tok::at:
2795 nextToken();
2796 if (FormatTok->is(tok::l_brace)) {
2797 nextToken();
2798 parseBracedList();
2799 }
2800 break;
2801 case tok::comma:
2802 SeenComma = true;
2803 nextToken();
2804 break;
2805 case tok::ellipsis:
2806 MightBeFoldExpr = true;
2807 nextToken();
2808 break;
2809 case tok::equal:
2810 SeenEqual = true;
2811 if (Style.isCSharp() && FormatTok->is(TT_FatArrow))
2812 tryToParseChildBlock();
2813 else
2814 nextToken();
2815 break;
2816 case tok::kw_class:
2817 if (Style.isJavaScript())
2818 parseRecord(/*ParseAsExpr=*/true);
2819 else
2820 nextToken();
2821 break;
2822 case tok::identifier:
2823 if (Style.isJavaScript() && (FormatTok->is(Keywords.kw_function)))
2824 tryToParseJSFunction();
2825 else
2826 nextToken();
2827 break;
2828 case tok::kw_switch:
2829 if (Style.isJava())
2830 parseSwitch(/*IsExpr=*/true);
2831 else
2832 nextToken();
2833 break;
2834 case tok::kw_requires:
2835 parseRequiresExpression();
2836 break;
2837 case tok::less:
2838 // We have here no clue whether this is a less, or a template opener, opt
2839 // out of the predefined StarAndAmpTokenType.
2840 ++ExcessLess;
2841 nextToken();
2842 break;
2843 case tok::greater:
2844 if (ExcessLess > 0)
2845 --ExcessLess;
2846 nextToken();
2847 break;
2848 case tok::star:
2849 case tok::amp:
2850 case tok::ampamp:
2851 if (StarAndAmpTokenType != TT_Unknown && ExcessLess == 0)
2852 FormatTok->setFinalizedType(StarAndAmpTokenType);
2853 [[fallthrough]];
2854 default:
2855 nextToken();
2856 break;
2857 }
2858 } while (!eof());
2859 return SeenEqual;
2860}
2861
2862void UnwrappedLineParser::parseSquare(bool LambdaIntroducer) {
2863 if (!LambdaIntroducer) {
2864 assert(FormatTok->is(tok::l_square) && "'[' expected.");
2865 if (tryToParseLambda())
2866 return;
2867 }
2868 do {
2869 switch (FormatTok->Tok.getKind()) {
2870 case tok::l_paren:
2871 parseParens();
2872 break;
2873 case tok::r_square:
2874 nextToken();
2875 return;
2876 case tok::r_brace:
2877 // A "}" inside parenthesis is an error if there wasn't a matching "{".
2878 return;
2879 case tok::l_square:
2880 parseSquare();
2881 break;
2882 case tok::l_brace: {
2883 if (!tryToParseBracedList())
2884 parseChildBlock();
2885 break;
2886 }
2887 case tok::at:
2888 case tok::colon:
2889 nextToken();
2890 if (FormatTok->is(tok::l_brace)) {
2891 nextToken();
2892 parseBracedList();
2893 }
2894 break;
2895 default:
2896 nextToken();
2897 break;
2898 }
2899 } while (!eof());
2900}
2901
2902void UnwrappedLineParser::keepAncestorBraces() {
2903 if (!Style.RemoveBracesLLVM)
2904 return;
2905
2906 const int MaxNestingLevels = 2;
2907 const int Size = NestedTooDeep.size();
2908 if (Size >= MaxNestingLevels)
2909 NestedTooDeep[Size - MaxNestingLevels] = true;
2910 NestedTooDeep.push_back(false);
2911}
2912
2914 for (const auto &Token : llvm::reverse(Line.Tokens))
2915 if (Token.Tok->isNot(tok::comment))
2916 return Token.Tok;
2917
2918 return nullptr;
2919}
2920
2921void UnwrappedLineParser::parseUnbracedBody(bool CheckEOF) {
2922 FormatToken *Tok = nullptr;
2923
2924 if (Style.InsertBraces && !Line->InPPDirective && !Line->Tokens.empty() &&
2925 PreprocessorDirectives.empty() && FormatTok->isNot(tok::semi)) {
2926 Tok = Style.BraceWrapping.AfterControlStatement == FormatStyle::BWACS_Never
2927 ? getLastNonComment(*Line)
2928 : Line->Tokens.back().Tok;
2929 assert(Tok);
2930 if (Tok->BraceCount < 0) {
2931 assert(Tok->BraceCount == -1);
2932 Tok = nullptr;
2933 } else {
2934 Tok->BraceCount = -1;
2935 }
2936 }
2937
2938 addUnwrappedLine();
2939 ++Line->Level;
2940 ++Line->UnbracedBodyLevel;
2941 parseStructuralElement();
2942 --Line->UnbracedBodyLevel;
2943
2944 if (Tok) {
2945 assert(!Line->InPPDirective);
2946 Tok = nullptr;
2947 for (const auto &L : llvm::reverse(*CurrentLines)) {
2948 if (!L.InPPDirective && getLastNonComment(L)) {
2949 Tok = L.Tokens.back().Tok;
2950 break;
2951 }
2952 }
2953 assert(Tok);
2954 ++Tok->BraceCount;
2955 }
2956
2957 if (CheckEOF && eof())
2958 addUnwrappedLine();
2959
2960 --Line->Level;
2961}
2962
2963static void markOptionalBraces(FormatToken *LeftBrace) {
2964 if (!LeftBrace)
2965 return;
2966
2967 assert(LeftBrace->is(tok::l_brace));
2968
2969 FormatToken *RightBrace = LeftBrace->MatchingParen;
2970 if (!RightBrace) {
2971 assert(!LeftBrace->Optional);
2972 return;
2973 }
2974
2975 assert(RightBrace->is(tok::r_brace));
2976 assert(RightBrace->MatchingParen == LeftBrace);
2977 assert(LeftBrace->Optional == RightBrace->Optional);
2978
2979 LeftBrace->Optional = true;
2980 RightBrace->Optional = true;
2981}
2982
2983void UnwrappedLineParser::handleAttributes() {
2984 // Handle AttributeMacro, e.g. `if (x) UNLIKELY`.
2985 if (FormatTok->isAttribute())
2986 nextToken();
2987 else if (FormatTok->is(tok::l_square))
2988 handleCppAttributes();
2989}
2990
2991bool UnwrappedLineParser::handleCppAttributes() {
2992 // Handle [[likely]] / [[unlikely]] attributes.
2993 assert(FormatTok->is(tok::l_square));
2994 if (!tryToParseSimpleAttribute())
2995 return false;
2996 parseSquare();
2997 return true;
2998}
2999
3000/// Returns whether \c Tok begins a block.
3001bool UnwrappedLineParser::isBlockBegin(const FormatToken &Tok) const {
3002 // FIXME: rename the function or make
3003 // Tok.isOneOf(tok::l_brace, TT_MacroBlockBegin) work.
3004 return Style.isVerilog() ? Keywords.isVerilogBegin(Tok)
3005 : Tok.is(tok::l_brace);
3006}
3007
3008FormatToken *UnwrappedLineParser::parseIfThenElse(IfStmtKind *IfKind,
3009 bool KeepBraces,
3010 bool IsVerilogAssert) {
3011 assert((FormatTok->is(tok::kw_if) ||
3012 (Style.isVerilog() &&
3013 FormatTok->isOneOf(tok::kw_restrict, Keywords.kw_assert,
3014 Keywords.kw_assume, Keywords.kw_cover))) &&
3015 "'if' expected");
3016 nextToken();
3017
3018 if (IsVerilogAssert) {
3019 // Handle `assert #0` and `assert final`.
3020 if (FormatTok->is(Keywords.kw_verilogHash)) {
3021 nextToken();
3022 if (FormatTok->is(tok::numeric_constant))
3023 nextToken();
3024 } else if (FormatTok->isOneOf(Keywords.kw_final, Keywords.kw_property,
3025 Keywords.kw_sequence)) {
3026 nextToken();
3027 }
3028 }
3029
3030 // TableGen's if statement has the form of `if <cond> then { ... }`.
3031 if (Style.isTableGen()) {
3032 while (!eof() && FormatTok->isNot(Keywords.kw_then)) {
3033 // Simply skip until then. This range only contains a value.
3034 nextToken();
3035 }
3036 }
3037
3038 // Handle `if !consteval`.
3039 if (FormatTok->is(tok::exclaim))
3040 nextToken();
3041
3042 bool KeepIfBraces = true;
3043 if (FormatTok->is(tok::kw_consteval)) {
3044 nextToken();
3045 } else {
3046 KeepIfBraces = !Style.RemoveBracesLLVM || KeepBraces;
3047 if (FormatTok->isOneOf(tok::kw_constexpr, tok::identifier))
3048 nextToken();
3049 if (FormatTok->is(tok::l_paren)) {
3050 FormatTok->setFinalizedType(TT_ConditionLParen);
3051 parseParens();
3052 }
3053 }
3054 handleAttributes();
3055 // The then action is optional in Verilog assert statements.
3056 if (IsVerilogAssert && FormatTok->is(tok::semi)) {
3057 nextToken();
3058 addUnwrappedLine();
3059 return nullptr;
3060 }
3061
3062 bool NeedsUnwrappedLine = false;
3063 keepAncestorBraces();
3064
3065 FormatToken *IfLeftBrace = nullptr;
3066 IfStmtKind IfBlockKind = IfStmtKind::NotIf;
3067
3068 if (isBlockBegin(*FormatTok)) {
3069 FormatTok->setFinalizedType(TT_ControlStatementLBrace);
3070 IfLeftBrace = FormatTok;
3071 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3072 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
3073 /*MunchSemi=*/true, KeepIfBraces, &IfBlockKind);
3074 setPreviousRBraceType(TT_ControlStatementRBrace);
3075 if (Style.BraceWrapping.BeforeElse)
3076 addUnwrappedLine();
3077 else
3078 NeedsUnwrappedLine = true;
3079 } else if (IsVerilogAssert && FormatTok->is(tok::kw_else)) {
3080 addUnwrappedLine();
3081 } else {
3082 parseUnbracedBody();
3083 }
3084
3085 if (Style.RemoveBracesLLVM) {
3086 assert(!NestedTooDeep.empty());
3087 KeepIfBraces = KeepIfBraces ||
3088 (IfLeftBrace && !IfLeftBrace->MatchingParen) ||
3089 NestedTooDeep.back() || IfBlockKind == IfStmtKind::IfOnly ||
3090 IfBlockKind == IfStmtKind::IfElseIf;
3091 }
3092
3093 bool KeepElseBraces = KeepIfBraces;
3094 FormatToken *ElseLeftBrace = nullptr;
3095 IfStmtKind Kind = IfStmtKind::IfOnly;
3096
3097 if (FormatTok->is(tok::kw_else)) {
3098 if (Style.RemoveBracesLLVM) {
3099 NestedTooDeep.back() = false;
3100 Kind = IfStmtKind::IfElse;
3101 }
3102 nextToken();
3103 handleAttributes();
3104 if (isBlockBegin(*FormatTok)) {
3105 const bool FollowedByIf = Tokens->peekNextToken()->is(tok::kw_if);
3106 FormatTok->setFinalizedType(TT_ElseLBrace);
3107 ElseLeftBrace = FormatTok;
3108 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3109 IfStmtKind ElseBlockKind = IfStmtKind::NotIf;
3110 FormatToken *IfLBrace =
3111 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
3112 /*MunchSemi=*/true, KeepElseBraces, &ElseBlockKind);
3113 setPreviousRBraceType(TT_ElseRBrace);
3114 if (FormatTok->is(tok::kw_else)) {
3115 KeepElseBraces = KeepElseBraces ||
3116 ElseBlockKind == IfStmtKind::IfOnly ||
3117 ElseBlockKind == IfStmtKind::IfElseIf;
3118 } else if (FollowedByIf && IfLBrace && !IfLBrace->Optional) {
3119 KeepElseBraces = true;
3120 assert(ElseLeftBrace->MatchingParen);
3121 markOptionalBraces(ElseLeftBrace);
3122 }
3123 addUnwrappedLine();
3124 } else if (!IsVerilogAssert && FormatTok->is(tok::kw_if)) {
3125 const FormatToken *Previous = Tokens->getPreviousToken();
3126 assert(Previous);
3127 const bool IsPrecededByComment = Previous->is(tok::comment);
3128 if (IsPrecededByComment) {
3129 addUnwrappedLine();
3130 ++Line->Level;
3131 }
3132 bool TooDeep = true;
3133 if (Style.RemoveBracesLLVM) {
3134 Kind = IfStmtKind::IfElseIf;
3135 TooDeep = NestedTooDeep.pop_back_val();
3136 }
3137 ElseLeftBrace = parseIfThenElse(/*IfKind=*/nullptr, KeepIfBraces);
3138 if (Style.RemoveBracesLLVM)
3139 NestedTooDeep.push_back(TooDeep);
3140 if (IsPrecededByComment)
3141 --Line->Level;
3142 } else {
3143 parseUnbracedBody(/*CheckEOF=*/true);
3144 }
3145 } else {
3146 KeepIfBraces = KeepIfBraces || IfBlockKind == IfStmtKind::IfElse;
3147 if (NeedsUnwrappedLine)
3148 addUnwrappedLine();
3149 }
3150
3151 if (!Style.RemoveBracesLLVM)
3152 return nullptr;
3153
3154 assert(!NestedTooDeep.empty());
3155 KeepElseBraces = KeepElseBraces ||
3156 (ElseLeftBrace && !ElseLeftBrace->MatchingParen) ||
3157 NestedTooDeep.back();
3158
3159 NestedTooDeep.pop_back();
3160
3161 if (!KeepIfBraces && !KeepElseBraces) {
3162 markOptionalBraces(IfLeftBrace);
3163 markOptionalBraces(ElseLeftBrace);
3164 } else if (IfLeftBrace) {
3165 FormatToken *IfRightBrace = IfLeftBrace->MatchingParen;
3166 if (IfRightBrace) {
3167 assert(IfRightBrace->MatchingParen == IfLeftBrace);
3168 assert(!IfLeftBrace->Optional);
3169 assert(!IfRightBrace->Optional);
3170 IfLeftBrace->MatchingParen = nullptr;
3171 IfRightBrace->MatchingParen = nullptr;
3172 }
3173 }
3174
3175 if (IfKind)
3176 *IfKind = Kind;
3177
3178 return IfLeftBrace;
3179}
3180
3181void UnwrappedLineParser::parseTryCatch() {
3182 assert(FormatTok->isOneOf(tok::kw_try, tok::kw___try) && "'try' expected");
3183 nextToken();
3184 bool NeedsUnwrappedLine = false;
3185 bool HasCtorInitializer = false;
3186 if (FormatTok->is(tok::colon)) {
3187 auto *Colon = FormatTok;
3188 // We are in a function try block, what comes is an initializer list.
3189 nextToken();
3190 if (FormatTok->is(tok::identifier)) {
3191 HasCtorInitializer = true;
3192 Colon->setFinalizedType(TT_CtorInitializerColon);
3193 }
3194
3195 // In case identifiers were removed by clang-tidy, what might follow is
3196 // multiple commas in sequence - before the first identifier.
3197 while (FormatTok->is(tok::comma))
3198 nextToken();
3199
3200 while (FormatTok->is(tok::identifier)) {
3201 nextToken();
3202 if (FormatTok->is(tok::l_paren)) {
3203 parseParens();
3204 } else if (FormatTok->is(tok::l_brace)) {
3205 nextToken();
3206 parseBracedList();
3207 }
3208
3209 // In case identifiers were removed by clang-tidy, what might follow is
3210 // multiple commas in sequence - after the first identifier.
3211 while (FormatTok->is(tok::comma))
3212 nextToken();
3213 }
3214 }
3215 // Parse try with resource.
3216 if (Style.isJava() && FormatTok->is(tok::l_paren))
3217 parseParens();
3218
3219 keepAncestorBraces();
3220
3221 if (FormatTok->is(tok::l_brace)) {
3222 if (HasCtorInitializer)
3223 FormatTok->setFinalizedType(TT_FunctionLBrace);
3224 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3225 parseBlock();
3226 if (Style.BraceWrapping.BeforeCatch)
3227 addUnwrappedLine();
3228 else
3229 NeedsUnwrappedLine = true;
3230 } else if (FormatTok->isNot(tok::kw_catch)) {
3231 // The C++ standard requires a compound-statement after a try.
3232 // If there's none, we try to assume there's a structuralElement
3233 // and try to continue.
3234 addUnwrappedLine();
3235 ++Line->Level;
3236 parseStructuralElement();
3237 --Line->Level;
3238 }
3239 for (bool SeenCatch = false;;) {
3240 if (FormatTok->is(tok::at))
3241 nextToken();
3242 if (FormatTok->isNoneOf(tok::kw_catch, Keywords.kw___except,
3243 tok::kw___finally, tok::objc_catch,
3244 tok::objc_finally) &&
3245 !((Style.isJava() || Style.isJavaScript()) &&
3246 FormatTok->is(Keywords.kw_finally))) {
3247 break;
3248 }
3249 if (FormatTok->is(tok::kw_catch))
3250 SeenCatch = true;
3251 nextToken();
3252 while (FormatTok->isNot(tok::l_brace)) {
3253 if (FormatTok->is(tok::l_paren)) {
3254 parseParens();
3255 continue;
3256 }
3257 if (FormatTok->isOneOf(tok::semi, tok::r_brace) || eof()) {
3258 if (Style.RemoveBracesLLVM)
3259 NestedTooDeep.pop_back();
3260 return;
3261 }
3262 nextToken();
3263 }
3264 if (SeenCatch) {
3265 FormatTok->setFinalizedType(TT_ControlStatementLBrace);
3266 SeenCatch = false;
3267 }
3268 NeedsUnwrappedLine = false;
3269 Line->MustBeDeclaration = false;
3270 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3271 parseBlock();
3272 if (Style.BraceWrapping.BeforeCatch)
3273 addUnwrappedLine();
3274 else
3275 NeedsUnwrappedLine = true;
3276 }
3277
3278 if (Style.RemoveBracesLLVM)
3279 NestedTooDeep.pop_back();
3280
3281 if (NeedsUnwrappedLine)
3282 addUnwrappedLine();
3283}
3284
3285void UnwrappedLineParser::parseNamespaceOrExportBlock(unsigned AddLevels) {
3286 bool ManageWhitesmithsBraces =
3287 AddLevels == 0u && Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths;
3288
3289 // If we're in Whitesmiths mode, indent the brace if we're not indenting
3290 // the whole block.
3291 if (ManageWhitesmithsBraces)
3292 ++Line->Level;
3293
3294 // Munch the semicolon after the block. This is more common than one would
3295 // think. Putting the semicolon into its own line is very ugly.
3296 parseBlock(/*MustBeDeclaration=*/true, AddLevels, /*MunchSemi=*/true,
3297 /*KeepBraces=*/true, /*IfKind=*/nullptr, ManageWhitesmithsBraces);
3298
3299 addUnwrappedLine(AddLevels > 0 ? LineLevel::Remove : LineLevel::Keep);
3300
3301 if (ManageWhitesmithsBraces)
3302 --Line->Level;
3303}
3304
3305void UnwrappedLineParser::parseNamespace() {
3306 assert(FormatTok->isOneOf(tok::kw_namespace, TT_NamespaceMacro) &&
3307 "'namespace' expected");
3308
3309 const FormatToken &InitialToken = *FormatTok;
3310 nextToken();
3311 if (InitialToken.is(TT_NamespaceMacro)) {
3312 parseParens();
3313 } else {
3314 while (FormatTok->isOneOf(tok::identifier, tok::coloncolon, tok::kw_inline,
3315 tok::l_square, tok::period, tok::l_paren) ||
3316 (Style.isCSharp() && FormatTok->is(tok::kw_union))) {
3317 if (FormatTok->is(tok::l_square))
3318 parseSquare();
3319 else if (FormatTok->is(tok::l_paren))
3320 parseParens();
3321 else
3322 nextToken();
3323 }
3324 }
3325 if (FormatTok->is(tok::l_brace)) {
3326 FormatTok->setFinalizedType(TT_NamespaceLBrace);
3327
3328 if (ShouldBreakBeforeBrace(Style, InitialToken,
3329 Tokens->peekNextToken()->is(tok::r_brace))) {
3330 addUnwrappedLine();
3331 }
3332
3333 unsigned AddLevels =
3334 Style.NamespaceIndentation == FormatStyle::NI_All ||
3335 (Style.NamespaceIndentation == FormatStyle::NI_Inner &&
3336 DeclarationScopeStack.size() > 1)
3337 ? 1u
3338 : 0u;
3339 parseNamespaceOrExportBlock(AddLevels);
3340 }
3341 // FIXME: Add error handling.
3342}
3343
3344void UnwrappedLineParser::parseCppExportBlock() {
3345 if (FormatTok->is(tok::l_brace)) {
3346 FormatTok->setFinalizedType(TT_ExportLBrace);
3347 if (Style.BraceWrapping.AfterExportBlock)
3348 addUnwrappedLine();
3349 }
3350 parseNamespaceOrExportBlock(/*AddLevels=*/Style.IndentExportBlock ? 1 : 0);
3351}
3352
3353void UnwrappedLineParser::parseNew() {
3354 assert(FormatTok->is(tok::kw_new) && "'new' expected");
3355 nextToken();
3356
3357 if (Style.isCSharp()) {
3358 do {
3359 // Handle constructor invocation, e.g. `new(field: value)`.
3360 if (FormatTok->is(tok::l_paren))
3361 parseParens();
3362
3363 // Handle array initialization syntax, e.g. `new[] {10, 20, 30}`.
3364 if (FormatTok->is(tok::l_brace))
3365 parseBracedList();
3366
3367 if (FormatTok->isOneOf(tok::semi, tok::comma))
3368 return;
3369
3370 nextToken();
3371 } while (!eof());
3372 }
3373
3374 if (!Style.isJava())
3375 return;
3376
3377 // In Java, we can parse everything up to the parens, which aren't optional.
3378 do {
3379 // There should not be a ;, { or } before the new's open paren.
3380 if (FormatTok->isOneOf(tok::semi, tok::l_brace, tok::r_brace))
3381 return;
3382
3383 // Consume the parens.
3384 if (FormatTok->is(tok::l_paren)) {
3385 parseParens();
3386
3387 // If there is a class body of an anonymous class, consume that as child.
3388 if (FormatTok->is(tok::l_brace))
3389 parseChildBlock();
3390 return;
3391 }
3392 nextToken();
3393 } while (!eof());
3394}
3395
3396void UnwrappedLineParser::parseLoopBody(bool KeepBraces, bool WrapRightBrace) {
3397 keepAncestorBraces();
3398
3399 if (isBlockBegin(*FormatTok)) {
3400 FormatTok->setFinalizedType(TT_ControlStatementLBrace);
3401 FormatToken *LeftBrace = FormatTok;
3402 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3403 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
3404 /*MunchSemi=*/true, KeepBraces);
3405 setPreviousRBraceType(TT_ControlStatementRBrace);
3406 if (!KeepBraces) {
3407 assert(!NestedTooDeep.empty());
3408 if (!NestedTooDeep.back())
3409 markOptionalBraces(LeftBrace);
3410 }
3411 if (WrapRightBrace)
3412 addUnwrappedLine();
3413 } else {
3414 parseUnbracedBody();
3415 }
3416
3417 if (!KeepBraces)
3418 NestedTooDeep.pop_back();
3419}
3420
3421void UnwrappedLineParser::parseForOrWhileLoop(bool HasParens) {
3422 assert((FormatTok->isOneOf(tok::kw_for, tok::kw_while, TT_ForEachMacro) ||
3423 (Style.isVerilog() &&
3424 FormatTok->isOneOf(Keywords.kw_always, Keywords.kw_always_comb,
3425 Keywords.kw_always_ff, Keywords.kw_always_latch,
3426 Keywords.kw_final, Keywords.kw_initial,
3427 Keywords.kw_foreach, Keywords.kw_forever,
3428 Keywords.kw_repeat))) &&
3429 "'for', 'while' or foreach macro expected");
3430 const bool KeepBraces = !Style.RemoveBracesLLVM ||
3431 FormatTok->isNoneOf(tok::kw_for, tok::kw_while);
3432
3433 nextToken();
3434 // JS' for await ( ...
3435 if (Style.isJavaScript() && FormatTok->is(Keywords.kw_await))
3436 nextToken();
3437 if (IsCpp && FormatTok->is(tok::kw_co_await))
3438 nextToken();
3439 if (HasParens && FormatTok->is(tok::l_paren)) {
3440 // The type is only set for Verilog basically because we were afraid to
3441 // change the existing behavior for loops. See the discussion on D121756 for
3442 // details.
3443 if (Style.isVerilog())
3444 FormatTok->setFinalizedType(TT_ConditionLParen);
3445 parseParens();
3446 }
3447
3448 if (Style.isVerilog()) {
3449 // Event control.
3450 parseVerilogSensitivityList();
3451 } else if (Style.AllowShortLoopsOnASingleLine && FormatTok->is(tok::semi) &&
3452 Tokens->getPreviousToken()->is(tok::r_paren)) {
3453 nextToken();
3454 addUnwrappedLine();
3455 return;
3456 }
3457
3458 handleAttributes();
3459 parseLoopBody(KeepBraces, /*WrapRightBrace=*/true);
3460}
3461
3462void UnwrappedLineParser::parseDoWhile() {
3463 assert(FormatTok->is(tok::kw_do) && "'do' expected");
3464 nextToken();
3465
3466 parseLoopBody(/*KeepBraces=*/true, Style.BraceWrapping.BeforeWhile);
3467
3468 // FIXME: Add error handling.
3469 if (FormatTok->isNot(tok::kw_while)) {
3470 addUnwrappedLine();
3471 return;
3472 }
3473
3474 FormatTok->setFinalizedType(TT_DoWhile);
3475
3476 // If in Whitesmiths mode, the line with the while() needs to be indented
3477 // to the same level as the block.
3478 if (Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths)
3479 ++Line->Level;
3480
3481 nextToken();
3482 parseStructuralElement();
3483}
3484
3485void UnwrappedLineParser::parseLabel(bool IsGotoLabel) {
3486 nextToken();
3487
3488 const auto IndentGotoLabel = Style.IndentGotoLabels;
3489 const auto OldLineLevel = Line->Level;
3490 auto &Level = Line->Level;
3491
3492 if (IsGotoLabel && IndentGotoLabel == FormatStyle::IGLS_NoIndent)
3493 Level = 0;
3494
3495 if (!IsGotoLabel || IndentGotoLabel == FormatStyle::IGLS_OuterIndent) {
3496 if (OldLineLevel > 1 || (!Line->InPPDirective && OldLineLevel > 0))
3497 --Level;
3498 }
3499
3500 if (!IsGotoLabel && !Style.IndentCaseBlocks &&
3501 CommentsBeforeNextToken.empty() && FormatTok->is(tok::l_brace)) {
3502 CompoundStatementIndenter Indenter(this, Level,
3503 Style.BraceWrapping.AfterCaseLabel,
3504 Style.BraceWrapping.IndentBraces);
3505 parseBlock();
3506 if (FormatTok->is(tok::kw_break)) {
3507 if (Style.BraceWrapping.AfterControlStatement ==
3509 addUnwrappedLine();
3510 if (!Style.IndentCaseBlocks &&
3511 Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths) {
3512 ++Level;
3513 }
3514 }
3515 parseStructuralElement();
3516 }
3517 addUnwrappedLine();
3518 } else {
3519 if (FormatTok->is(tok::semi))
3520 nextToken();
3521 addUnwrappedLine();
3522 }
3523
3524 Level = OldLineLevel;
3525
3526 if (FormatTok->isNot(tok::l_brace)) {
3527 parseStructuralElement();
3528 addUnwrappedLine();
3529 }
3530}
3531
3532void UnwrappedLineParser::parseCaseLabel() {
3533 assert(FormatTok->is(tok::kw_case) && "'case' expected");
3534 auto *Case = FormatTok;
3535
3536 // FIXME: fix handling of complex expressions here.
3537 do {
3538 nextToken();
3539 if (FormatTok->is(tok::colon)) {
3540 FormatTok->setFinalizedType(TT_CaseLabelColon);
3541 break;
3542 }
3543 if (Style.isJava() && FormatTok->is(tok::arrow)) {
3544 FormatTok->setFinalizedType(TT_CaseLabelArrow);
3545 Case->setFinalizedType(TT_SwitchExpressionLabel);
3546 break;
3547 }
3548 } while (!eof());
3549 parseLabel();
3550}
3551
3552void UnwrappedLineParser::parseSwitch(bool IsExpr) {
3553 assert(FormatTok->is(tok::kw_switch) && "'switch' expected");
3554 nextToken();
3555 if (FormatTok->is(tok::l_paren))
3556 parseParens();
3557
3558 keepAncestorBraces();
3559
3560 if (FormatTok->is(tok::l_brace)) {
3561 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3562 FormatTok->setFinalizedType(IsExpr ? TT_SwitchExpressionLBrace
3563 : TT_ControlStatementLBrace);
3564 if (IsExpr)
3565 parseChildBlock();
3566 else
3567 parseBlock();
3568 setPreviousRBraceType(TT_ControlStatementRBrace);
3569 if (!IsExpr)
3570 addUnwrappedLine();
3571 } else {
3572 addUnwrappedLine();
3573 ++Line->Level;
3574 parseStructuralElement();
3575 --Line->Level;
3576 }
3577
3578 if (Style.RemoveBracesLLVM)
3579 NestedTooDeep.pop_back();
3580}
3581
3582void UnwrappedLineParser::parseAccessSpecifier() {
3583 nextToken();
3584 // Understand Qt's slots.
3585 if (FormatTok->isOneOf(Keywords.kw_slots, Keywords.kw_qslots))
3586 nextToken();
3587 // Otherwise, we don't know what it is, and we'd better keep the next token.
3588 if (FormatTok->is(tok::colon))
3589 nextToken();
3590 addUnwrappedLine();
3591}
3592
3593/// Parses a requires, decides if it is a clause or an expression.
3594/// \pre The current token has to be the requires keyword.
3595/// \returns true if it parsed a clause.
3596bool UnwrappedLineParser::parseRequires(bool SeenEqual) {
3597 assert(FormatTok->is(tok::kw_requires) && "'requires' expected");
3598
3599 // We try to guess if it is a requires clause, or a requires expression. For
3600 // that we first check the next token.
3601 switch (Tokens->peekNextToken(/*SkipComment=*/true)->Tok.getKind()) {
3602 case tok::l_brace:
3603 // This can only be an expression, never a clause.
3604 parseRequiresExpression();
3605 return false;
3606 case tok::l_paren:
3607 // Clauses and expression can start with a paren, it's unclear what we have.
3608 break;
3609 default:
3610 // All other tokens can only be a clause.
3611 parseRequiresClause();
3612 return true;
3613 }
3614
3615 // Looking forward we would have to decide if there are function declaration
3616 // like arguments to the requires expression:
3617 // requires (T t) {
3618 // Or there is a constraint expression for the requires clause:
3619 // requires (C<T> && ...
3620
3621 // But first let's look behind.
3622 auto *PreviousNonComment = FormatTok->getPreviousNonComment();
3623
3624 if (!PreviousNonComment ||
3625 PreviousNonComment->is(TT_RequiresExpressionLBrace)) {
3626 // If there is no token, or an expression left brace, we are a requires
3627 // clause within a requires expression.
3628 parseRequiresClause();
3629 return true;
3630 }
3631
3632 switch (PreviousNonComment->Tok.getKind()) {
3633 case tok::greater:
3634 case tok::r_paren:
3635 case tok::kw_noexcept:
3636 case tok::kw_const:
3637 case tok::star:
3638 case tok::amp:
3639 // This is a requires clause.
3640 parseRequiresClause();
3641 return true;
3642 case tok::ampamp: {
3643 // This can be either:
3644 // if (... && requires (T t) ...)
3645 // Or
3646 // void member(...) && requires (C<T> ...
3647 // We check the one token before that for a const:
3648 // void member(...) const && requires (C<T> ...
3649 auto PrevPrev = PreviousNonComment->getPreviousNonComment();
3650 if ((PrevPrev && PrevPrev->is(tok::kw_const)) || !SeenEqual) {
3651 parseRequiresClause();
3652 return true;
3653 }
3654 break;
3655 }
3656 default:
3657 if (PreviousNonComment->isTypeOrIdentifier(LangOpts)) {
3658 // This is a requires clause.
3659 parseRequiresClause();
3660 return true;
3661 }
3662 // It's an expression.
3663 parseRequiresExpression();
3664 return false;
3665 }
3666
3667 // Now we look forward and try to check if the paren content is a parameter
3668 // list. The parameters can be cv-qualified and contain references or
3669 // pointers.
3670 // So we want basically to check for TYPE NAME, but TYPE can contain all kinds
3671 // of stuff: typename, const, *, &, &&, ::, identifiers.
3672
3673 unsigned StoredPosition = Tokens->getPosition();
3674 FormatToken *NextToken = Tokens->getNextToken();
3675 int Lookahead = 0;
3676 auto PeekNext = [&Lookahead, &NextToken, this] {
3677 ++Lookahead;
3678 NextToken = Tokens->getNextToken();
3679 };
3680
3681 bool FoundType = false;
3682 bool LastWasColonColon = false;
3683 int OpenAngles = 0;
3684
3685 for (; Lookahead < 50; PeekNext()) {
3686 switch (NextToken->Tok.getKind()) {
3687 case tok::kw_volatile:
3688 case tok::kw_const:
3689 case tok::comma:
3690 if (OpenAngles == 0) {
3691 FormatTok = Tokens->setPosition(StoredPosition);
3692 parseRequiresExpression();
3693 return false;
3694 }
3695 break;
3696 case tok::eof:
3697 // Break out of the loop.
3698 Lookahead = 50;
3699 break;
3700 case tok::coloncolon:
3701 LastWasColonColon = true;
3702 break;
3703 case tok::kw_decltype:
3704 case tok::identifier:
3705 if (FoundType && !LastWasColonColon && OpenAngles == 0) {
3706 FormatTok = Tokens->setPosition(StoredPosition);
3707 parseRequiresExpression();
3708 return false;
3709 }
3710 FoundType = true;
3711 LastWasColonColon = false;
3712 break;
3713 case tok::less:
3714 ++OpenAngles;
3715 break;
3716 case tok::greater:
3717 --OpenAngles;
3718 break;
3719 default:
3720 if (NextToken->isTypeName(LangOpts)) {
3721 FormatTok = Tokens->setPosition(StoredPosition);
3722 parseRequiresExpression();
3723 return false;
3724 }
3725 break;
3726 }
3727 }
3728 // This seems to be a complicated expression, just assume it's a clause.
3729 FormatTok = Tokens->setPosition(StoredPosition);
3730 parseRequiresClause();
3731 return true;
3732}
3733
3734/// Parses a requires clause.
3735/// \sa parseRequiresExpression
3736///
3737/// Returns if it either has finished parsing the clause, or it detects, that
3738/// the clause is incorrect.
3739void UnwrappedLineParser::parseRequiresClause() {
3740 assert(FormatTok->is(tok::kw_requires) && "'requires' expected");
3741
3742 // If there is no previous token, we are within a requires expression,
3743 // otherwise we will always have the template or function declaration in front
3744 // of it.
3745 bool InRequiresExpression =
3746 !FormatTok->Previous ||
3747 FormatTok->Previous->is(TT_RequiresExpressionLBrace);
3748
3749 FormatTok->setFinalizedType(InRequiresExpression
3750 ? TT_RequiresClauseInARequiresExpression
3751 : TT_RequiresClause);
3752 nextToken();
3753
3754 // NOTE: parseConstraintExpression is only ever called from this function.
3755 // It could be inlined into here.
3756 parseConstraintExpression();
3757
3758 if (!InRequiresExpression && FormatTok->Previous)
3759 FormatTok->Previous->ClosesRequiresClause = true;
3760}
3761
3762/// Parses a requires expression.
3763/// \sa parseRequiresClause
3764///
3765/// Returns if it either has finished parsing the expression, or it detects,
3766/// that the expression is incorrect.
3767void UnwrappedLineParser::parseRequiresExpression() {
3768 assert(FormatTok->is(tok::kw_requires) && "'requires' expected");
3769
3770 FormatTok->setFinalizedType(TT_RequiresExpression);
3771 nextToken();
3772
3773 if (FormatTok->is(tok::l_paren)) {
3774 FormatTok->setFinalizedType(TT_RequiresExpressionLParen);
3775 parseParens();
3776 }
3777
3778 if (FormatTok->is(tok::l_brace)) {
3779 FormatTok->setFinalizedType(TT_RequiresExpressionLBrace);
3780 parseChildBlock();
3781 }
3782}
3783
3784/// Parses a constraint expression.
3785///
3786/// This is the body of a requires clause. It returns, when the parsing is
3787/// complete, or the expression is incorrect.
3788void UnwrappedLineParser::parseConstraintExpression() {
3789 // The special handling for lambdas is needed since tryToParseLambda() eats a
3790 // token and if a requires expression is the last part of a requires clause
3791 // and followed by an attribute like [[nodiscard]] the ClosesRequiresClause is
3792 // not set on the correct token. Thus we need to be aware if we even expect a
3793 // lambda to be possible.
3794 // template <typename T> requires requires { ... } [[nodiscard]] ...;
3795 bool LambdaNextTimeAllowed = true;
3796
3797 // Within lambda declarations, it is permitted to put a requires clause after
3798 // its template parameter list, which would place the requires clause right
3799 // before the parentheses of the parameters of the lambda declaration. Thus,
3800 // we track if we expect to see grouping parentheses at all.
3801 // Without this check, `requires foo<T> (T t)` in the below example would be
3802 // seen as the whole requires clause, accidentally eating the parameters of
3803 // the lambda.
3804 // [&]<typename T> requires foo<T> (T t) { ... };
3805 bool TopLevelParensAllowed = true;
3806
3807 do {
3808 bool LambdaThisTimeAllowed = std::exchange(LambdaNextTimeAllowed, false);
3809
3810 switch (FormatTok->Tok.getKind()) {
3811 case tok::kw_requires:
3812 parseRequiresExpression();
3813 break;
3814
3815 case tok::l_paren:
3816 if (!TopLevelParensAllowed)
3817 return;
3818 parseParens(/*AmpAmpTokenType=*/TT_BinaryOperator);
3819 TopLevelParensAllowed = false;
3820 break;
3821
3822 case tok::l_square:
3823 if (!LambdaThisTimeAllowed || !tryToParseLambda())
3824 return;
3825 break;
3826
3827 case tok::kw_const:
3828 case tok::semi:
3829 case tok::kw_class:
3830 case tok::kw_struct:
3831 case tok::kw_union:
3832 return;
3833
3834 case tok::l_brace:
3835 // Potential function body.
3836 return;
3837
3838 case tok::ampamp:
3839 case tok::pipepipe:
3840 FormatTok->setFinalizedType(TT_BinaryOperator);
3841 nextToken();
3842 LambdaNextTimeAllowed = true;
3843 TopLevelParensAllowed = true;
3844 break;
3845
3846 case tok::comma:
3847 case tok::comment:
3848 LambdaNextTimeAllowed = LambdaThisTimeAllowed;
3849 nextToken();
3850 break;
3851
3852 case tok::kw_sizeof:
3853 case tok::greater:
3854 case tok::greaterequal:
3855 case tok::greatergreater:
3856 case tok::less:
3857 case tok::lessequal:
3858 case tok::lessless:
3859 case tok::equalequal:
3860 case tok::exclaim:
3861 case tok::exclaimequal:
3862 case tok::plus:
3863 case tok::minus:
3864 case tok::star:
3865 case tok::slash:
3866 LambdaNextTimeAllowed = true;
3867 TopLevelParensAllowed = true;
3868 // Just eat them.
3869 nextToken();
3870 break;
3871
3872 case tok::numeric_constant:
3873 case tok::coloncolon:
3874 case tok::kw_true:
3875 case tok::kw_false:
3876 TopLevelParensAllowed = false;
3877 // Just eat them.
3878 nextToken();
3879 break;
3880
3881 case tok::kw_static_cast:
3882 case tok::kw_const_cast:
3883 case tok::kw_reinterpret_cast:
3884 case tok::kw_dynamic_cast:
3885 nextToken();
3886 if (FormatTok->isNot(tok::less))
3887 return;
3888
3889 nextToken();
3890 parseBracedList(/*IsAngleBracket=*/true);
3891 break;
3892
3893 default:
3894 if (!FormatTok->Tok.getIdentifierInfo()) {
3895 // Identifiers are part of the default case, we check for more then
3896 // tok::identifier to handle builtin type traits.
3897 return;
3898 }
3899
3900 // We need to differentiate identifiers for a template deduction guide,
3901 // variables, or function return types (the constraint expression has
3902 // ended before that), and basically all other cases. But it's easier to
3903 // check the other way around.
3904 assert(FormatTok->Previous);
3905 switch (FormatTok->Previous->Tok.getKind()) {
3906 case tok::coloncolon: // Nested identifier.
3907 case tok::ampamp: // Start of a function or variable for the
3908 case tok::pipepipe: // constraint expression. (binary)
3909 case tok::exclaim: // The same as above, but unary.
3910 case tok::kw_requires: // Initial identifier of a requires clause.
3911 case tok::equal: // Initial identifier of a concept declaration.
3912 case tok::kw_template: // A dependent template.
3913 break;
3914 default:
3915 return;
3916 }
3917
3918 // Read identifier with optional template declaration.
3919 nextToken();
3920 if (FormatTok->is(tok::less)) {
3921 nextToken();
3922 parseBracedList(/*IsAngleBracket=*/true);
3923 }
3924 TopLevelParensAllowed = false;
3925 break;
3926 }
3927 } while (!eof());
3928}
3929
3930bool UnwrappedLineParser::parseEnum() {
3931 const FormatToken &InitialToken = *FormatTok;
3932
3933 // Won't be 'enum' for NS_ENUMs.
3934 if (FormatTok->is(tok::kw_enum))
3935 nextToken();
3936
3937 // In TypeScript, "enum" can also be used as property name, e.g. in interface
3938 // declarations. An "enum" keyword followed by a colon would be a syntax
3939 // error and thus assume it is just an identifier.
3940 if (Style.isJavaScript() && FormatTok->isOneOf(tok::colon, tok::question))
3941 return false;
3942
3943 // In protobuf, "enum" can be used as a field name.
3944 if (Style.Language == FormatStyle::LK_Proto && FormatTok->is(tok::equal))
3945 return false;
3946
3947 if (IsCpp) {
3948 // Eat up enum class ...
3949 if (FormatTok->isOneOf(tok::kw_class, tok::kw_struct))
3950 nextToken();
3951 while (FormatTok->is(tok::l_square))
3952 if (!handleCppAttributes())
3953 return false;
3954 }
3955
3956 while (FormatTok->Tok.getIdentifierInfo() ||
3957 FormatTok->isOneOf(tok::colon, tok::coloncolon, tok::less,
3958 tok::greater, tok::comma, tok::question,
3959 tok::l_square)) {
3960 if (FormatTok->is(tok::colon))
3961 FormatTok->setFinalizedType(TT_EnumUnderlyingTypeColon);
3962 if (Style.isVerilog()) {
3963 FormatTok->setFinalizedType(TT_VerilogDimensionedTypeName);
3964 nextToken();
3965 // In Verilog the base type can have dimensions.
3966 while (FormatTok->is(tok::l_square))
3967 parseSquare();
3968 } else {
3969 nextToken();
3970 }
3971 // We can have macros or attributes in between 'enum' and the enum name.
3972 if (FormatTok->is(tok::l_paren))
3973 parseParens();
3974 if (FormatTok->is(tok::identifier)) {
3975 nextToken();
3976 // If there are two identifiers in a row, this is likely an elaborate
3977 // return type. In Java, this can be "implements", etc.
3978 if (IsCpp && FormatTok->is(tok::identifier))
3979 return false;
3980 }
3981 }
3982
3983 // Just a declaration or something is wrong.
3984 if (FormatTok->isNot(tok::l_brace))
3985 return true;
3986 FormatTok->setFinalizedType(TT_EnumLBrace);
3987 FormatTok->setBlockKind(BK_Block);
3988
3989 if (Style.isJava()) {
3990 // Java enums are different.
3991 parseJavaEnumBody();
3992 return true;
3993 }
3994 if (Style.Language == FormatStyle::LK_Proto) {
3995 parseBlock(/*MustBeDeclaration=*/true);
3996 return true;
3997 }
3998
3999 const bool ManageWhitesmithsBraces =
4000 Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths;
4001
4002 if (!Style.AllowShortEnumsOnASingleLine &&
4003 ShouldBreakBeforeBrace(Style, InitialToken,
4004 Tokens->peekNextToken()->is(tok::r_brace))) {
4005 addUnwrappedLine();
4006
4007 // If we're in Whitesmiths mode, indent the brace if we're not indenting
4008 // the whole block.
4009 if (ManageWhitesmithsBraces)
4010 ++Line->Level;
4011 }
4012 // Parse enum body.
4013 nextToken();
4014 if (!Style.AllowShortEnumsOnASingleLine) {
4015 addUnwrappedLine();
4016 if (!ManageWhitesmithsBraces)
4017 ++Line->Level;
4018 }
4019 const auto OpeningLineIndex = CurrentLines->empty()
4020 ? UnwrappedLine::kInvalidIndex
4021 : CurrentLines->size() - 1;
4022 bool HasError = !parseBracedList(/*IsAngleBracket=*/false, /*IsEnum=*/true);
4023 if (!Style.AllowShortEnumsOnASingleLine && !ManageWhitesmithsBraces)
4024 --Line->Level;
4025 if (HasError) {
4026 if (FormatTok->is(tok::semi))
4027 nextToken();
4028 addUnwrappedLine();
4029 }
4030 setPreviousRBraceType(TT_EnumRBrace);
4031 if (ManageWhitesmithsBraces)
4032 Line->MatchingOpeningBlockLineIndex = OpeningLineIndex;
4033 return true;
4034
4035 // There is no addUnwrappedLine() here so that we fall through to parsing a
4036 // structural element afterwards. Thus, in "enum A {} n, m;",
4037 // "} n, m;" will end up in one unwrapped line.
4038}
4039
4040bool UnwrappedLineParser::parseStructLike() {
4041 // parseRecord falls through and does not yet add an unwrapped line as a
4042 // record declaration or definition can start a structural element.
4043 parseRecord();
4044 // This does not apply to Java, JavaScript and C#.
4045 if (Style.isJava() || Style.isJavaScript() || Style.isCSharp()) {
4046 if (FormatTok->is(tok::semi))
4047 nextToken();
4048 addUnwrappedLine();
4049 return true;
4050 }
4051 return false;
4052}
4053
4054namespace {
4055// A class used to set and restore the Token position when peeking
4056// ahead in the token source.
4057class ScopedTokenPosition {
4058 unsigned StoredPosition;
4059 FormatTokenSource *Tokens;
4060
4061public:
4062 ScopedTokenPosition(FormatTokenSource *Tokens) : Tokens(Tokens) {
4063 assert(Tokens && "Tokens expected to not be null");
4064 StoredPosition = Tokens->getPosition();
4065 }
4066
4067 ~ScopedTokenPosition() { Tokens->setPosition(StoredPosition); }
4068};
4069} // namespace
4070
4071// Look to see if we have [[ by looking ahead, if
4072// its not then rewind to the original position.
4073bool UnwrappedLineParser::tryToParseSimpleAttribute() {
4074 ScopedTokenPosition AutoPosition(Tokens);
4075 FormatToken *Tok = Tokens->getNextToken();
4076 // We already read the first [ check for the second.
4077 if (Tok->isNot(tok::l_square))
4078 return false;
4079 // Double check that the attribute is just something
4080 // fairly simple.
4081 while (Tok->isNot(tok::eof)) {
4082 if (Tok->is(tok::r_square))
4083 break;
4084 Tok = Tokens->getNextToken();
4085 }
4086 if (Tok->is(tok::eof))
4087 return false;
4088 Tok = Tokens->getNextToken();
4089 if (Tok->isNot(tok::r_square))
4090 return false;
4091 Tok = Tokens->getNextToken();
4092 if (Tok->is(tok::semi))
4093 return false;
4094 return true;
4095}
4096
4097void UnwrappedLineParser::parseJavaEnumBody() {
4098 assert(FormatTok->is(tok::l_brace));
4099 const FormatToken *OpeningBrace = FormatTok;
4100
4101 // Determine whether the enum is simple, i.e. does not have a semicolon or
4102 // constants with class bodies. Simple enums can be formatted like braced
4103 // lists, contracted to a single line, etc.
4104 unsigned StoredPosition = Tokens->getPosition();
4105 bool IsSimple = true;
4106 FormatToken *Tok = Tokens->getNextToken();
4107 while (Tok->isNot(tok::eof)) {
4108 if (Tok->is(tok::r_brace))
4109 break;
4110 if (Tok->isOneOf(tok::l_brace, tok::semi)) {
4111 IsSimple = false;
4112 break;
4113 }
4114 // FIXME: This will also mark enums with braces in the arguments to enum
4115 // constants as "not simple". This is probably fine in practice, though.
4116 Tok = Tokens->getNextToken();
4117 }
4118 FormatTok = Tokens->setPosition(StoredPosition);
4119
4120 if (IsSimple) {
4121 nextToken();
4122 parseBracedList();
4123 addUnwrappedLine();
4124 return;
4125 }
4126
4127 // Parse the body of a more complex enum.
4128 // First add a line for everything up to the "{".
4129 nextToken();
4130 addUnwrappedLine();
4131 ++Line->Level;
4132
4133 // Parse the enum constants.
4134 while (!eof()) {
4135 if (FormatTok->is(tok::l_brace)) {
4136 // Parse the constant's class body.
4137 parseBlock(/*MustBeDeclaration=*/true, /*AddLevels=*/1u,
4138 /*MunchSemi=*/false);
4139 } else if (FormatTok->is(tok::l_paren)) {
4140 parseParens();
4141 } else if (FormatTok->is(tok::comma)) {
4142 nextToken();
4143 addUnwrappedLine();
4144 } else if (FormatTok->is(tok::semi)) {
4145 nextToken();
4146 addUnwrappedLine();
4147 break;
4148 } else if (FormatTok->is(tok::r_brace)) {
4149 addUnwrappedLine();
4150 break;
4151 } else {
4152 nextToken();
4153 }
4154 }
4155
4156 // Parse the class body after the enum's ";" if any.
4157 parseLevel(OpeningBrace);
4158 nextToken();
4159 --Line->Level;
4160 addUnwrappedLine();
4161}
4162
4163void UnwrappedLineParser::parseRecord(bool ParseAsExpr, bool IsJavaRecord) {
4164 assert(!IsJavaRecord || FormatTok->is(Keywords.kw_record));
4165 const FormatToken &InitialToken = *FormatTok;
4166 nextToken();
4167
4168 FormatToken *ClassName =
4169 IsJavaRecord && FormatTok->is(tok::identifier) ? FormatTok : nullptr;
4170 bool IsDerived = false;
4171 auto IsNonMacroIdentifier = [](const FormatToken *Tok) {
4172 return Tok->is(tok::identifier) && Tok->TokenText != Tok->TokenText.upper();
4173 };
4174 // JavaScript/TypeScript supports anonymous classes like:
4175 // a = class extends foo { }
4176 bool JSPastExtendsOrImplements = false;
4177 // The actual identifier can be a nested name specifier, and in macros
4178 // it is often token-pasted.
4179 // An [[attribute]] can be before the identifier.
4180 while (FormatTok->isOneOf(tok::identifier, tok::coloncolon, tok::hashhash,
4181 tok::kw_alignas, tok::l_square) ||
4182 FormatTok->isAttribute() ||
4183 ((Style.isJava() || Style.isJavaScript()) &&
4184 FormatTok->isOneOf(tok::period, tok::comma)) ||
4185 (Style.isVerilog() &&
4186 FormatTok->isOneOf(tok::kw_signed, tok::kw_unsigned))) {
4187 if (Style.isJavaScript() &&
4188 FormatTok->isOneOf(Keywords.kw_extends, Keywords.kw_implements)) {
4189 JSPastExtendsOrImplements = true;
4190 // JavaScript/TypeScript supports inline object types in
4191 // extends/implements positions:
4192 // class Foo implements {bar: number} { }
4193 nextToken();
4194 if (FormatTok->is(tok::l_brace)) {
4195 tryToParseBracedList();
4196 continue;
4197 }
4198 }
4199 if (FormatTok->is(tok::l_square) && handleCppAttributes())
4200 continue;
4201 auto *Previous = FormatTok;
4202 nextToken();
4203 switch (FormatTok->Tok.getKind()) {
4204 case tok::l_paren:
4205 // We can have macros in between 'class' and the class name.
4206 if (IsJavaRecord || !IsNonMacroIdentifier(Previous) ||
4207 // e.g. `struct macro(a) S { int i; };`
4208 Previous->Previous == &InitialToken) {
4209 parseParens();
4210 }
4211 break;
4212 case tok::coloncolon:
4213 case tok::hashhash:
4214 break;
4215 default:
4216 if (JSPastExtendsOrImplements || ClassName ||
4217 Previous->isNot(tok::identifier) || Previous->is(TT_AttributeMacro)) {
4218 break;
4219 }
4220 if (const auto Text = Previous->TokenText;
4221 Text.size() == 1 || Text != Text.upper()) {
4222 ClassName = Previous;
4223 }
4224 }
4225 }
4226
4227 auto IsListInitialization = [&] {
4228 if (!ClassName || IsDerived || JSPastExtendsOrImplements)
4229 return false;
4230 assert(FormatTok->is(tok::l_brace));
4231 const auto *Prev = FormatTok->getPreviousNonComment();
4232 assert(Prev);
4233 return Prev != ClassName && Prev->is(tok::identifier) &&
4234 Prev->isNot(Keywords.kw_final) && tryToParseBracedList();
4235 };
4236
4237 if (FormatTok->isOneOf(tok::colon, tok::less)) {
4238 int AngleNestingLevel = 0;
4239 do {
4240 if (FormatTok->is(tok::less))
4241 ++AngleNestingLevel;
4242 else if (FormatTok->is(tok::greater))
4243 --AngleNestingLevel;
4244
4245 if (AngleNestingLevel == 0) {
4246 if (FormatTok->is(tok::colon)) {
4247 IsDerived = true;
4248 } else if (!IsDerived && FormatTok->is(tok::identifier) &&
4249 FormatTok->Previous->is(tok::coloncolon)) {
4250 ClassName = FormatTok;
4251 } else if (FormatTok->is(tok::l_paren) &&
4252 IsNonMacroIdentifier(FormatTok->Previous)) {
4253 break;
4254 }
4255 }
4256 if (FormatTok->is(tok::l_brace)) {
4257 if (AngleNestingLevel == 0 && IsListInitialization())
4258 return;
4259 calculateBraceTypes(/*ExpectClassBody=*/true);
4260 if (!tryToParseBracedList())
4261 break;
4262 }
4263 if (FormatTok->is(tok::l_square)) {
4264 FormatToken *Previous = FormatTok->Previous;
4265 if (!Previous || (Previous->isNot(tok::r_paren) &&
4266 !Previous->isTypeOrIdentifier(LangOpts))) {
4267 // Don't try parsing a lambda if we had a closing parenthesis before,
4268 // it was probably a pointer to an array: int (*)[].
4269 if (!tryToParseLambda())
4270 continue;
4271 } else {
4272 parseSquare();
4273 continue;
4274 }
4275 }
4276 if (FormatTok->is(tok::semi))
4277 return;
4278 if (Style.isCSharp() && FormatTok->is(Keywords.kw_where)) {
4279 addUnwrappedLine();
4280 nextToken();
4281 parseCSharpGenericTypeConstraint();
4282 break;
4283 }
4284 nextToken();
4285 } while (!eof());
4286 }
4287
4288 auto GetBraceTypes =
4289 [](const FormatToken &RecordTok) -> std::pair<TokenType, TokenType> {
4290 switch (RecordTok.Tok.getKind()) {
4291 case tok::kw_class:
4292 return {TT_ClassLBrace, TT_ClassRBrace};
4293 case tok::kw_struct:
4294 return {TT_StructLBrace, TT_StructRBrace};
4295 case tok::kw_union:
4296 return {TT_UnionLBrace, TT_UnionRBrace};
4297 default:
4298 // Useful for e.g. interface.
4299 return {TT_RecordLBrace, TT_RecordRBrace};
4300 }
4301 };
4302 if (FormatTok->is(tok::l_brace)) {
4303 if (IsListInitialization())
4304 return;
4305 if (ClassName)
4306 ClassName->setFinalizedType(TT_ClassHeadName);
4307 auto [OpenBraceType, ClosingBraceType] = GetBraceTypes(InitialToken);
4308 FormatTok->setFinalizedType(OpenBraceType);
4309 if (ParseAsExpr) {
4310 parseChildBlock();
4311 } else {
4312 if (ShouldBreakBeforeBrace(Style, InitialToken,
4313 Tokens->peekNextToken()->is(tok::r_brace),
4314 IsJavaRecord)) {
4315 addUnwrappedLine();
4316 }
4317
4318 bool IndentAfterExplicitAccessModifier = false;
4319 unsigned AddLevels = 1u;
4320 switch (Style.IndentAccessModifiers) {
4322 break;
4324 if (Style.isCpp()) {
4325 IndentAfterExplicitAccessModifier = true;
4326 break;
4327 }
4328 // Other languages use the same indentation as IAMS_Always.
4329 [[fallthrough]];
4331 AddLevels = 2u;
4332 break;
4333 }
4334 parseBlock(/*MustBeDeclaration=*/true, AddLevels, /*MunchSemi=*/false,
4335 /*KeepBraces=*/true, /*IfKind=*/nullptr,
4336 /*UnindentWhitesmithsBraces=*/false,
4337 IndentAfterExplicitAccessModifier);
4338 }
4339 setPreviousRBraceType(ClosingBraceType);
4340 }
4341 // There is no addUnwrappedLine() here so that we fall through to parsing a
4342 // structural element afterwards. Thus, in "class A {} n, m;",
4343 // "} n, m;" will end up in one unwrapped line.
4344}
4345
4346void UnwrappedLineParser::parseObjCMethod() {
4347 assert(FormatTok->isOneOf(tok::l_paren, tok::identifier) &&
4348 "'(' or identifier expected.");
4349 do {
4350 if (FormatTok->is(tok::semi)) {
4351 nextToken();
4352 addUnwrappedLine();
4353 return;
4354 } else if (FormatTok->is(tok::l_brace)) {
4355 if (Style.BraceWrapping.AfterFunction)
4356 addUnwrappedLine();
4357 parseBlock();
4358 addUnwrappedLine();
4359 return;
4360 } else {
4361 nextToken();
4362 }
4363 } while (!eof());
4364}
4365
4366void UnwrappedLineParser::parseObjCProtocolList() {
4367 assert(FormatTok->is(tok::less) && "'<' expected.");
4368 do {
4369 nextToken();
4370 // Early exit in case someone forgot a close angle.
4371 if (FormatTok->isOneOf(tok::semi, tok::l_brace, tok::objc_end))
4372 return;
4373 } while (!eof() && FormatTok->isNot(tok::greater));
4374 nextToken(); // Skip '>'.
4375}
4376
4377void UnwrappedLineParser::parseObjCUntilAtEnd() {
4378 do {
4379 if (FormatTok->is(tok::objc_end)) {
4380 nextToken();
4381 addUnwrappedLine();
4382 break;
4383 }
4384 if (FormatTok->is(tok::l_brace)) {
4385 parseBlock();
4386 // In ObjC interfaces, nothing should be following the "}".
4387 addUnwrappedLine();
4388 } else if (FormatTok->is(tok::r_brace)) {
4389 // Ignore stray "}". parseStructuralElement doesn't consume them.
4390 nextToken();
4391 addUnwrappedLine();
4392 } else if (FormatTok->isOneOf(tok::minus, tok::plus)) {
4393 nextToken();
4394 if (FormatTok->isOneOf(tok::l_paren, tok::identifier))
4395 parseObjCMethod();
4396 } else {
4397 parseStructuralElement();
4398 }
4399 } while (!eof());
4400}
4401
4402void UnwrappedLineParser::parseObjCInterfaceOrImplementation() {
4403 assert(FormatTok->isOneOf(tok::objc_interface, tok::objc_implementation));
4404 nextToken();
4405 nextToken(); // interface name
4406
4407 // @interface can be followed by a lightweight generic
4408 // specialization list, then either a base class or a category.
4409 if (FormatTok->is(tok::less))
4410 parseObjCLightweightGenerics();
4411 if (FormatTok->is(tok::colon)) {
4412 nextToken();
4413 nextToken(); // base class name
4414 // The base class can also have lightweight generics applied to it.
4415 if (FormatTok->is(tok::less))
4416 parseObjCLightweightGenerics();
4417 } else if (FormatTok->is(tok::l_paren)) {
4418 // Skip category, if present.
4419 parseParens();
4420 }
4421
4422 if (FormatTok->is(tok::less))
4423 parseObjCProtocolList();
4424
4425 if (FormatTok->is(tok::l_brace)) {
4426 if (Style.BraceWrapping.AfterObjCDeclaration)
4427 addUnwrappedLine();
4428 parseBlock(/*MustBeDeclaration=*/true);
4429 }
4430
4431 // With instance variables, this puts '}' on its own line. Without instance
4432 // variables, this ends the @interface line.
4433 addUnwrappedLine();
4434
4435 parseObjCUntilAtEnd();
4436}
4437
4438void UnwrappedLineParser::parseObjCLightweightGenerics() {
4439 assert(FormatTok->is(tok::less));
4440 // Unlike protocol lists, generic parameterizations support
4441 // nested angles:
4442 //
4443 // @interface Foo<ValueType : id <NSCopying, NSSecureCoding>> :
4444 // NSObject <NSCopying, NSSecureCoding>
4445 //
4446 // so we need to count how many open angles we have left.
4447 unsigned NumOpenAngles = 1;
4448 do {
4449 nextToken();
4450 // Early exit in case someone forgot a close angle.
4451 if (FormatTok->isOneOf(tok::semi, tok::l_brace, tok::objc_end))
4452 break;
4453 if (FormatTok->is(tok::less)) {
4454 ++NumOpenAngles;
4455 } else if (FormatTok->is(tok::greater)) {
4456 assert(NumOpenAngles > 0 && "'>' makes NumOpenAngles negative");
4457 --NumOpenAngles;
4458 }
4459 } while (!eof() && NumOpenAngles != 0);
4460 nextToken(); // Skip '>'.
4461}
4462
4463// Returns true for the declaration/definition form of @protocol,
4464// false for the expression form.
4465bool UnwrappedLineParser::parseObjCProtocol() {
4466 assert(FormatTok->is(tok::objc_protocol));
4467 nextToken();
4468
4469 if (FormatTok->is(tok::l_paren)) {
4470 // The expression form of @protocol, e.g. "Protocol* p = @protocol(foo);".
4471 return false;
4472 }
4473
4474 // The definition/declaration form,
4475 // @protocol Foo
4476 // - (int)someMethod;
4477 // @end
4478
4479 nextToken(); // protocol name
4480
4481 if (FormatTok->is(tok::less))
4482 parseObjCProtocolList();
4483
4484 // Check for protocol declaration.
4485 if (FormatTok->is(tok::semi)) {
4486 nextToken();
4487 addUnwrappedLine();
4488 return true;
4489 }
4490
4491 addUnwrappedLine();
4492 parseObjCUntilAtEnd();
4493 return true;
4494}
4495
4496void UnwrappedLineParser::parseJavaScriptEs6ImportExport() {
4497 bool IsImport = FormatTok->is(Keywords.kw_import);
4498 assert(IsImport || FormatTok->is(tok::kw_export));
4499 nextToken();
4500
4501 // Consume the "default" in "export default class/function".
4502 if (FormatTok->is(tok::kw_default))
4503 nextToken();
4504
4505 // Consume "async function", "function" and "default function", so that these
4506 // get parsed as free-standing JS functions, i.e. do not require a trailing
4507 // semicolon.
4508 if (FormatTok->is(Keywords.kw_async))
4509 nextToken();
4510 if (FormatTok->is(Keywords.kw_function)) {
4511 nextToken();
4512 return;
4513 }
4514
4515 // For imports, `export *`, `export {...}`, consume the rest of the line up
4516 // to the terminating `;`. For everything else, just return and continue
4517 // parsing the structural element, i.e. the declaration or expression for
4518 // `export default`.
4519 if (!IsImport && FormatTok->isNoneOf(tok::l_brace, tok::star) &&
4520 !FormatTok->isStringLiteral() &&
4521 !(FormatTok->is(Keywords.kw_type) &&
4522 Tokens->peekNextToken()->isOneOf(tok::l_brace, tok::star))) {
4523 return;
4524 }
4525
4526 while (!eof()) {
4527 if (FormatTok->is(tok::semi))
4528 return;
4529 if (Line->Tokens.empty()) {
4530 // Common issue: Automatic Semicolon Insertion wrapped the line, so the
4531 // import statement should terminate.
4532 return;
4533 }
4534 if (FormatTok->is(tok::l_brace)) {
4535 FormatTok->setBlockKind(BK_Block);
4536 nextToken();
4537 parseBracedList();
4538 } else {
4539 nextToken();
4540 }
4541 }
4542}
4543
4544void UnwrappedLineParser::parseStatementMacro() {
4545 nextToken();
4546 if (FormatTok->is(tok::l_paren))
4547 parseParens();
4548 if (FormatTok->is(tok::semi))
4549 nextToken();
4550 addUnwrappedLine();
4551}
4552
4553void UnwrappedLineParser::parseVerilogHierarchyIdentifier() {
4554 // consume things like a::`b.c[d:e] or a::*
4555 while (true) {
4556 if (FormatTok->isOneOf(tok::star, tok::period, tok::periodstar,
4557 tok::coloncolon, tok::hash) ||
4558 Keywords.isVerilogIdentifier(*FormatTok)) {
4559 nextToken();
4560 } else if (FormatTok->is(tok::l_square)) {
4561 parseSquare();
4562 } else {
4563 break;
4564 }
4565 }
4566}
4567
4568void UnwrappedLineParser::parseVerilogSensitivityList() {
4569 if (FormatTok->isNot(tok::at))
4570 return;
4571 nextToken();
4572 // A block event expression has 2 at signs.
4573 if (FormatTok->is(tok::at))
4574 nextToken();
4575 switch (FormatTok->Tok.getKind()) {
4576 case tok::star:
4577 nextToken();
4578 break;
4579 case tok::l_paren:
4580 parseParens();
4581 break;
4582 default:
4583 parseVerilogHierarchyIdentifier();
4584 break;
4585 }
4586}
4587
4588unsigned UnwrappedLineParser::parseVerilogHierarchyHeader() {
4589 unsigned AddLevels = 0;
4590
4591 if (FormatTok->is(Keywords.kw_clocking)) {
4592 nextToken();
4593 if (Keywords.isVerilogIdentifier(*FormatTok))
4594 nextToken();
4595 parseVerilogSensitivityList();
4596 if (FormatTok->is(tok::semi))
4597 nextToken();
4598 } else if (FormatTok->isOneOf(tok::kw_case, Keywords.kw_casex,
4599 Keywords.kw_casez, Keywords.kw_randcase,
4600 Keywords.kw_randsequence)) {
4601 if (Style.IndentCaseLabels)
4602 AddLevels++;
4603 nextToken();
4604 if (FormatTok->is(tok::l_paren)) {
4605 FormatTok->setFinalizedType(TT_ConditionLParen);
4606 parseParens();
4607 }
4608 if (FormatTok->isOneOf(Keywords.kw_inside, Keywords.kw_matches))
4609 nextToken();
4610 // The case header has no semicolon.
4611 } else {
4612 // "module" etc.
4613 nextToken();
4614 // all the words like the name of the module and specifiers like
4615 // "automatic" and the width of function return type
4616 while (true) {
4617 if (FormatTok->is(tok::l_square)) {
4618 auto Prev = FormatTok->getPreviousNonComment();
4619 if (Prev && Keywords.isVerilogIdentifier(*Prev))
4620 Prev->setFinalizedType(TT_VerilogDimensionedTypeName);
4621 parseSquare();
4622 } else if (Keywords.isVerilogIdentifier(*FormatTok) ||
4623 FormatTok->isOneOf(tok::hash, tok::hashhash, tok::coloncolon,
4624 Keywords.kw_automatic, tok::kw_static)) {
4625 nextToken();
4626 } else {
4627 break;
4628 }
4629 }
4630
4631 auto NewLine = [this]() {
4632 addUnwrappedLine();
4633 Line->IsContinuation = true;
4634 };
4635
4636 // package imports
4637 while (FormatTok->is(Keywords.kw_import)) {
4638 NewLine();
4639 nextToken();
4640 parseVerilogHierarchyIdentifier();
4641 if (FormatTok->is(tok::semi))
4642 nextToken();
4643 }
4644
4645 // parameters and ports
4646 if (FormatTok->is(Keywords.kw_verilogHash)) {
4647 NewLine();
4648 nextToken();
4649 if (FormatTok->is(tok::l_paren)) {
4650 FormatTok->setFinalizedType(TT_VerilogMultiLineListLParen);
4651 parseParens();
4652 }
4653 }
4654 if (FormatTok->is(tok::l_paren)) {
4655 NewLine();
4656 FormatTok->setFinalizedType(TT_VerilogMultiLineListLParen);
4657 parseParens();
4658 }
4659
4660 // extends and implements
4661 if (FormatTok->is(Keywords.kw_extends)) {
4662 NewLine();
4663 nextToken();
4664 parseVerilogHierarchyIdentifier();
4665 if (FormatTok->is(tok::l_paren))
4666 parseParens();
4667 }
4668 if (FormatTok->is(Keywords.kw_implements)) {
4669 NewLine();
4670 do {
4671 nextToken();
4672 parseVerilogHierarchyIdentifier();
4673 } while (FormatTok->is(tok::comma));
4674 }
4675
4676 // Coverage event for cover groups.
4677 if (FormatTok->is(tok::at)) {
4678 NewLine();
4679 parseVerilogSensitivityList();
4680 }
4681
4682 if (FormatTok->is(tok::semi))
4683 nextToken(/*LevelDifference=*/1);
4684 addUnwrappedLine();
4685 }
4686
4687 return AddLevels;
4688}
4689
4690void UnwrappedLineParser::parseVerilogTable() {
4691 assert(FormatTok->is(Keywords.kw_table));
4692 nextToken(/*LevelDifference=*/1);
4693 addUnwrappedLine();
4694
4695 auto InitialLevel = Line->Level++;
4696 while (!eof() && !Keywords.isVerilogEnd(*FormatTok)) {
4697 FormatToken *Tok = FormatTok;
4698 nextToken();
4699 if (Tok->is(tok::semi))
4700 addUnwrappedLine();
4701 else if (Tok->isOneOf(tok::star, tok::colon, tok::question, tok::minus))
4702 Tok->setFinalizedType(TT_VerilogTableItem);
4703 }
4704 Line->Level = InitialLevel;
4705 nextToken(/*LevelDifference=*/-1);
4706 addUnwrappedLine();
4707}
4708
4709void UnwrappedLineParser::parseVerilogCaseLabel() {
4710 // The label will get unindented in AnnotatingParser. If there are no leading
4711 // spaces, indent the rest here so that things inside the block will be
4712 // indented relative to things outside. We don't use parseLabel because we
4713 // don't know whether this colon is a label or a ternary expression at this
4714 // point.
4715 auto OrigLevel = Line->Level;
4716 auto FirstLine = CurrentLines->size();
4717 if (Line->Level == 0 || (Line->InPPDirective && Line->Level <= 1))
4718 ++Line->Level;
4719 else if (!Style.IndentCaseBlocks && Keywords.isVerilogBegin(*FormatTok))
4720 --Line->Level;
4721 parseStructuralElement();
4722 // Restore the indentation in both the new line and the line that has the
4723 // label.
4724 if (CurrentLines->size() > FirstLine)
4725 (*CurrentLines)[FirstLine].Level = OrigLevel;
4726 Line->Level = OrigLevel;
4727}
4728
4729void UnwrappedLineParser::parseVerilogExtern() {
4730 assert(
4731 FormatTok->isOneOf(tok::kw_extern, tok::kw_export, Keywords.kw_import));
4732 nextToken();
4733 // "DPI-C"
4734 if (FormatTok->is(tok::string_literal))
4735 nextToken();
4736 skipVerilogQualifiers();
4737 if (Keywords.isVerilogIdentifier(*FormatTok))
4738 nextToken();
4739 if (FormatTok->is(tok::equal))
4740 nextToken();
4741 if (Keywords.isVerilogHierarchy(*FormatTok))
4742 parseVerilogHierarchyHeader();
4743}
4744
4745void UnwrappedLineParser::skipVerilogQualifiers() {
4746 while (FormatTok->isOneOf(tok::kw_protected, tok::kw_virtual, tok::kw_static,
4747 Keywords.kw_rand, Keywords.kw_context,
4748 Keywords.kw_pure, Keywords.kw_randc,
4749 Keywords.kw_local)) {
4750 nextToken();
4751 }
4752}
4753
4754bool UnwrappedLineParser::containsExpansion(const UnwrappedLine &Line) const {
4755 for (const auto &N : Line.Tokens) {
4756 if (N.Tok->MacroCtx)
4757 return true;
4758 for (const UnwrappedLine &Child : N.Children)
4759 if (containsExpansion(Child))
4760 return true;
4761 }
4762 return false;
4763}
4764
4765void UnwrappedLineParser::addUnwrappedLine(LineLevel AdjustLevel) {
4766 if (Line->Tokens.empty())
4767 return;
4768 LLVM_DEBUG({
4769 if (!parsingPPDirective()) {
4770 llvm::dbgs() << "Adding unwrapped line:\n";
4771 printDebugInfo(*Line);
4772 }
4773 });
4774
4775 // If this line closes a block when in Whitesmiths mode, remember that
4776 // information so that the level can be decreased after the line is added.
4777 // This has to happen after the addition of the line since the line itself
4778 // needs to be indented.
4779 bool ClosesWhitesmithsBlock =
4780 Line->MatchingOpeningBlockLineIndex != UnwrappedLine::kInvalidIndex &&
4781 Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths;
4782
4783 // If the current line was expanded from a macro call, we use it to
4784 // reconstruct an unwrapped line from the structure of the expanded unwrapped
4785 // line and the unexpanded token stream.
4786 if (!parsingPPDirective() && !InExpansion && containsExpansion(*Line)) {
4787 if (!Reconstruct)
4788 Reconstruct.emplace(Line->Level, Unexpanded);
4789 Reconstruct->addLine(*Line);
4790
4791 // While the reconstructed unexpanded lines are stored in the normal
4792 // flow of lines, the expanded lines are stored on the side to be analyzed
4793 // in an extra step.
4794 CurrentExpandedLines.push_back(std::move(*Line));
4795
4796 if (Reconstruct->finished()) {
4797 UnwrappedLine Reconstructed = std::move(*Reconstruct).takeResult();
4798 assert(!Reconstructed.Tokens.empty() &&
4799 "Reconstructed must at least contain the macro identifier.");
4800 assert(!parsingPPDirective());
4801 LLVM_DEBUG({
4802 llvm::dbgs() << "Adding unexpanded line:\n";
4803 printDebugInfo(Reconstructed);
4804 });
4805 ExpandedLines[Reconstructed.Tokens.begin()->Tok] = CurrentExpandedLines;
4806 Lines.push_back(std::move(Reconstructed));
4807 CurrentExpandedLines.clear();
4808 Reconstruct.reset();
4809 }
4810 } else {
4811 // At the top level we only get here when no unexpansion is going on, or
4812 // when conditional formatting led to unfinished macro reconstructions.
4813 assert(!Reconstruct || (CurrentLines != &Lines) || !PP.Stack.empty());
4814 CurrentLines->push_back(std::move(*Line));
4815 }
4816 Line->Tokens.clear();
4817 Line->MatchingOpeningBlockLineIndex = UnwrappedLine::kInvalidIndex;
4818 Line->FirstStartColumn = 0;
4819 Line->IsContinuation = false;
4820 Line->SeenDecltypeAuto = false;
4821 Line->IsModuleOrImportDecl = false;
4822
4823 if (ClosesWhitesmithsBlock && AdjustLevel == LineLevel::Remove)
4824 --Line->Level;
4825 if (!parsingPPDirective() && !PreprocessorDirectives.empty()) {
4826 CurrentLines->append(
4827 std::make_move_iterator(PreprocessorDirectives.begin()),
4828 std::make_move_iterator(PreprocessorDirectives.end()));
4829 PreprocessorDirectives.clear();
4830 }
4831 // Disconnect the current token from the last token on the previous line.
4832 FormatTok->Previous = nullptr;
4833}
4834
4835bool UnwrappedLineParser::eof() const { return FormatTok->is(tok::eof); }
4836
4837bool UnwrappedLineParser::isOnNewLine(const FormatToken &FormatTok) {
4838 return (Line->InPPDirective || FormatTok.HasUnescapedNewline) &&
4839 FormatTok.NewlinesBefore > 0;
4840}
4841
4842// Checks if \p FormatTok is a line comment that continues the line comment
4843// section on \p Line.
4844static bool
4846 const UnwrappedLine &Line, const FormatStyle &Style,
4847 const llvm::Regex &CommentPragmasRegex) {
4848 if (Line.Tokens.empty() || Style.ReflowComments != FormatStyle::RCS_Always)
4849 return false;
4850
4851 StringRef IndentContent = FormatTok.TokenText;
4852 if (FormatTok.TokenText.starts_with("//") ||
4853 FormatTok.TokenText.starts_with("/*")) {
4854 IndentContent = FormatTok.TokenText.substr(2);
4855 }
4856 if (CommentPragmasRegex.match(IndentContent))
4857 return false;
4858
4859 // If Line starts with a line comment, then FormatTok continues the comment
4860 // section if its original column is greater or equal to the original start
4861 // column of the line.
4862 //
4863 // Define the min column token of a line as follows: if a line ends in '{' or
4864 // contains a '{' followed by a line comment, then the min column token is
4865 // that '{'. Otherwise, the min column token of the line is the first token of
4866 // the line.
4867 //
4868 // If Line starts with a token other than a line comment, then FormatTok
4869 // continues the comment section if its original column is greater than the
4870 // original start column of the min column token of the line.
4871 //
4872 // For example, the second line comment continues the first in these cases:
4873 //
4874 // // first line
4875 // // second line
4876 //
4877 // and:
4878 //
4879 // // first line
4880 // // second line
4881 //
4882 // and:
4883 //
4884 // int i; // first line
4885 // // second line
4886 //
4887 // and:
4888 //
4889 // do { // first line
4890 // // second line
4891 // int i;
4892 // } while (true);
4893 //
4894 // and:
4895 //
4896 // enum {
4897 // a, // first line
4898 // // second line
4899 // b
4900 // };
4901 //
4902 // The second line comment doesn't continue the first in these cases:
4903 //
4904 // // first line
4905 // // second line
4906 //
4907 // and:
4908 //
4909 // int i; // first line
4910 // // second line
4911 //
4912 // and:
4913 //
4914 // do { // first line
4915 // // second line
4916 // int i;
4917 // } while (true);
4918 //
4919 // and:
4920 //
4921 // enum {
4922 // a, // first line
4923 // // second line
4924 // };
4925 const FormatToken *MinColumnToken = Line.Tokens.front().Tok;
4926
4927 // Scan for '{//'. If found, use the column of '{' as a min column for line
4928 // comment section continuation.
4929 const FormatToken *PreviousToken = nullptr;
4930 for (const UnwrappedLineNode &Node : Line.Tokens) {
4931 if (PreviousToken && PreviousToken->is(tok::l_brace) &&
4932 isLineComment(*Node.Tok)) {
4933 MinColumnToken = PreviousToken;
4934 break;
4935 }
4936 PreviousToken = Node.Tok;
4937
4938 // Grab the last newline preceding a token in this unwrapped line.
4939 if (Node.Tok->NewlinesBefore > 0)
4940 MinColumnToken = Node.Tok;
4941 }
4942 if (PreviousToken && PreviousToken->is(tok::l_brace))
4943 MinColumnToken = PreviousToken;
4944
4945 return continuesLineComment(FormatTok, /*Previous=*/Line.Tokens.back().Tok,
4946 MinColumnToken);
4947}
4948
4949void UnwrappedLineParser::flushComments(bool NewlineBeforeNext) {
4950 bool JustComments = Line->Tokens.empty();
4951 for (FormatToken *Tok : CommentsBeforeNextToken) {
4952 // Line comments that belong to the same line comment section are put on the
4953 // same line since later we might want to reflow content between them.
4954 // Additional fine-grained breaking of line comment sections is controlled
4955 // by the class BreakableLineCommentSection in case it is desirable to keep
4956 // several line comment sections in the same unwrapped line.
4957 //
4958 // FIXME: Consider putting separate line comment sections as children to the
4959 // unwrapped line instead.
4960 Tok->ContinuesLineCommentSection =
4961 continuesLineCommentSection(*Tok, *Line, Style, CommentPragmasRegex);
4962 if (isOnNewLine(*Tok) && JustComments && !Tok->ContinuesLineCommentSection)
4963 addUnwrappedLine();
4964 pushToken(Tok);
4965 }
4966 if (NewlineBeforeNext && JustComments)
4967 addUnwrappedLine();
4968 CommentsBeforeNextToken.clear();
4969}
4970
4971void UnwrappedLineParser::nextToken(int LevelDifference) {
4972 if (eof())
4973 return;
4974 flushComments(isOnNewLine(*FormatTok));
4975 pushToken(FormatTok);
4976 FormatToken *Previous = FormatTok;
4977 if (!Style.isJavaScript())
4978 readToken(LevelDifference);
4979 else
4980 readTokenWithJavaScriptASI();
4981 FormatTok->Previous = Previous;
4982 if (Style.isVerilog()) {
4983 // Blocks in Verilog can have `begin` and `end` instead of braces. For
4984 // keywords like `begin`, we can't treat them the same as left braces
4985 // because some contexts require one of them. For example structs use
4986 // braces and if blocks use keywords, and a left brace can occur in an if
4987 // statement, but it is not a block. For keywords like `end`, we simply
4988 // treat them the same as right braces.
4989 if (Keywords.isVerilogEnd(*FormatTok))
4990 FormatTok->Tok.setKind(tok::r_brace);
4991 }
4992}
4993
4994void UnwrappedLineParser::distributeComments(
4995 const ArrayRef<FormatToken *> &Comments, const FormatToken *NextTok) {
4996 // Whether or not a line comment token continues a line is controlled by
4997 // the method continuesLineCommentSection, with the following caveat:
4998 //
4999 // Define a trail of Comments to be a nonempty proper postfix of Comments such
5000 // that each comment line from the trail is aligned with the next token, if
5001 // the next token exists. If a trail exists, the beginning of the maximal
5002 // trail is marked as a start of a new comment section.
5003 //
5004 // For example in this code:
5005 //
5006 // int a; // line about a
5007 // // line 1 about b
5008 // // line 2 about b
5009 // int b;
5010 //
5011 // the two lines about b form a maximal trail, so there are two sections, the
5012 // first one consisting of the single comment "// line about a" and the
5013 // second one consisting of the next two comments.
5014 if (Comments.empty())
5015 return;
5016 bool ShouldPushCommentsInCurrentLine = true;
5017 bool HasTrailAlignedWithNextToken = false;
5018 unsigned StartOfTrailAlignedWithNextToken = 0;
5019 if (NextTok) {
5020 // We are skipping the first element intentionally.
5021 for (unsigned i = Comments.size() - 1; i > 0; --i) {
5022 if (Comments[i]->OriginalColumn == NextTok->OriginalColumn) {
5023 HasTrailAlignedWithNextToken = true;
5024 StartOfTrailAlignedWithNextToken = i;
5025 }
5026 }
5027 }
5028 for (unsigned i = 0, e = Comments.size(); i < e; ++i) {
5029 FormatToken *FormatTok = Comments[i];
5030 if (HasTrailAlignedWithNextToken && i == StartOfTrailAlignedWithNextToken) {
5031 FormatTok->ContinuesLineCommentSection = false;
5032 } else {
5033 FormatTok->ContinuesLineCommentSection = continuesLineCommentSection(
5034 *FormatTok, *Line, Style, CommentPragmasRegex);
5035 }
5036 if (!FormatTok->ContinuesLineCommentSection &&
5037 (isOnNewLine(*FormatTok) || FormatTok->IsFirst)) {
5038 ShouldPushCommentsInCurrentLine = false;
5039 }
5040 if (ShouldPushCommentsInCurrentLine)
5041 pushToken(FormatTok);
5042 else
5043 CommentsBeforeNextToken.push_back(FormatTok);
5044 }
5045}
5046
5047void UnwrappedLineParser::readToken(int LevelDifference) {
5049 bool PreviousWasComment = false;
5050 bool FirstNonCommentOnLine = false;
5051 do {
5052 FormatTok = Tokens->getNextToken();
5053 assert(FormatTok);
5054 while (FormatTok->isOneOf(TT_ConflictStart, TT_ConflictEnd,
5055 TT_ConflictAlternative)) {
5056 if (FormatTok->is(TT_ConflictStart))
5057 conditionalCompilationStart(/*Unreachable=*/false);
5058 else if (FormatTok->is(TT_ConflictAlternative))
5059 conditionalCompilationAlternative();
5060 else if (FormatTok->is(TT_ConflictEnd))
5061 conditionalCompilationEnd();
5062 FormatTok = Tokens->getNextToken();
5063 FormatTok->MustBreakBefore = true;
5064 FormatTok->MustBreakBeforeFinalized = true;
5065 }
5066
5067 auto IsFirstNonCommentOnLine = [](bool FirstNonCommentOnLine,
5068 const FormatToken &Tok,
5069 bool PreviousWasComment) {
5070 auto IsFirstOnLine = [](const FormatToken &Tok) {
5071 return Tok.HasUnescapedNewline || Tok.IsFirst;
5072 };
5073
5074 // Consider preprocessor directives preceded by block comments as first
5075 // on line.
5076 if (PreviousWasComment)
5077 return FirstNonCommentOnLine || IsFirstOnLine(Tok);
5078 return IsFirstOnLine(Tok);
5079 };
5080
5081 FirstNonCommentOnLine = IsFirstNonCommentOnLine(
5082 FirstNonCommentOnLine, *FormatTok, PreviousWasComment);
5083 PreviousWasComment = FormatTok->is(tok::comment);
5084
5085 while (!Line->InPPDirective && FormatTok->is(tok::hash) &&
5086 FirstNonCommentOnLine) {
5087 // In Verilog, the backtick is used for macro invocations. In TableGen,
5088 // the single hash is used for the paste operator.
5089 const auto *Next = Tokens->peekNextToken();
5090 if ((Style.isVerilog() && !Keywords.isVerilogPPDirective(*Next)) ||
5091 (Style.isTableGen() &&
5092 Next->isNoneOf(tok::kw_else, tok::pp_define, tok::pp_ifdef,
5093 tok::pp_ifndef, tok::pp_endif))) {
5094 break;
5095 }
5096 distributeComments(Comments, FormatTok);
5097 Comments.clear();
5098 // If the directive was parsed before the token stream was rewound (see
5099 // parseMacroCall()), its lines were kept. Parse it again only for its
5100 // effect on the preprocessor bookkeeping and discard the new lines.
5101 const bool ParsedBefore = !ParsedPPDirectives.insert(FormatTok).second;
5102 // If there is an unfinished unwrapped line, we flush the preprocessor
5103 // directives only after that unwrapped line was finished later.
5104 bool SwitchToPreprocessorLines = !Line->Tokens.empty();
5105 ScopedLineState BlockState(*this, SwitchToPreprocessorLines,
5106 /*DiscardLines=*/ParsedBefore);
5107 assert((LevelDifference >= 0 ||
5108 static_cast<unsigned>(-LevelDifference) <= Line->Level) &&
5109 "LevelDifference makes Line->Level negative");
5110 Line->Level += LevelDifference;
5111 // Comments stored before the preprocessor directive need to be output
5112 // before the preprocessor directive, at the same level as the
5113 // preprocessor directive, as we consider them to apply to the directive.
5114 if (Style.IndentPPDirectives == FormatStyle::PPDIS_BeforeHash &&
5115 PP.BranchLevel > 0) {
5116 Line->Level += PP.BranchLevel;
5117 }
5118 assert(Line->Level >= Line->UnbracedBodyLevel);
5119 Line->Level -= Line->UnbracedBodyLevel;
5120 flushComments(isOnNewLine(*FormatTok));
5121 const bool IsEndIf = Tokens->peekNextToken()->is(tok::pp_endif);
5122 parsePPDirective();
5123 PreviousWasComment = FormatTok->is(tok::comment);
5124 FirstNonCommentOnLine = IsFirstNonCommentOnLine(
5125 FirstNonCommentOnLine, *FormatTok, PreviousWasComment);
5126 // If the #endif of a potential include guard is the last thing in the
5127 // file, then we found an include guard.
5128 if (IsEndIf && PP.IncludeGuard == IG_Defined && PP.BranchLevel == -1 &&
5129 getIncludeGuardState(Style.IndentPPDirectives) == IG_Inited &&
5130 (eof() ||
5131 (PreviousWasComment &&
5132 Tokens->peekNextToken(/*SkipComment=*/true)->is(tok::eof)))) {
5133 PP.IncludeGuard = IG_Found;
5134 }
5135 }
5136
5137 if (!PP.Stack.empty() && (PP.Stack.back().Kind == PP_Unreachable) &&
5138 !Line->InPPDirective) {
5139 continue;
5140 }
5141
5142 if (FormatTok->is(tok::identifier) &&
5143 Macros.defined(FormatTok->TokenText) &&
5144 // FIXME: Allow expanding macros in preprocessor directives.
5145 !Line->InPPDirective) {
5146 FormatToken *ID = FormatTok;
5147 unsigned Position = Tokens->getPosition();
5148 // Parsing the arguments of the call may parse preprocessor directives,
5149 // which are parsed again if the token stream is rewound because the
5150 // arguments are discarded. The preprocessor bookkeeping is restored
5151 // whenever that happens.
5152 const auto SavedPPState = PP;
5153
5154 // To correctly parse the code, we need to replace the tokens of the macro
5155 // call with its expansion.
5156 auto PreCall = std::move(Line);
5157 Line.reset(new UnwrappedLine);
5158 bool OldInExpansion = InExpansion;
5159 InExpansion = true;
5160 // We parse the macro call into a new line.
5161 auto Args = parseMacroCall(SavedPPState);
5162 InExpansion = OldInExpansion;
5163 assert(Line->Tokens.front().Tok == ID);
5164 // And remember the unexpanded macro call tokens.
5165 auto UnexpandedLine = std::move(Line);
5166 // Reset to the old line.
5167 Line = std::move(PreCall);
5168
5169 LLVM_DEBUG({
5170 llvm::dbgs() << "Macro call: " << ID->TokenText << "(";
5171 if (Args) {
5172 llvm::dbgs() << "(";
5173 for (const auto &Arg : Args.value())
5174 for (const auto &T : Arg)
5175 llvm::dbgs() << T->TokenText << " ";
5176 llvm::dbgs() << ")";
5177 }
5178 llvm::dbgs() << "\n";
5179 });
5180 if (Macros.objectLike(ID->TokenText) && Args &&
5181 !Macros.hasArity(ID->TokenText, Args->size())) {
5182 // The macro is either
5183 // - object-like, but we got argumnets, or
5184 // - overloaded to be both object-like and function-like, but none of
5185 // the function-like arities match the number of arguments.
5186 // Thus, expand as object-like macro.
5187 LLVM_DEBUG(llvm::dbgs()
5188 << "Macro \"" << ID->TokenText
5189 << "\" not overloaded for arity " << Args->size()
5190 << "or not function-like, using object-like overload.");
5191 Args.reset();
5192 UnexpandedLine->Tokens.resize(1);
5193 Tokens->setPosition(Position);
5194 // Not nextToken(), which would push the stale FormatTok onto the line.
5195 FormatTok = Tokens->getNextToken();
5196 PP = SavedPPState;
5197 assert(!Args && Macros.objectLike(ID->TokenText));
5198 }
5199 if ((!Args && Macros.objectLike(ID->TokenText)) ||
5200 (Args && Macros.hasArity(ID->TokenText, Args->size()))) {
5201 // Next, we insert the expanded tokens in the token stream at the
5202 // current position, and continue parsing.
5203 Unexpanded[ID] = std::move(UnexpandedLine);
5205 Macros.expand(ID, std::move(Args));
5206 if (!Expansion.empty())
5207 FormatTok = Tokens->insertTokens(Expansion);
5208
5209 LLVM_DEBUG({
5210 llvm::dbgs() << "Expanded: ";
5211 for (const auto &T : Expansion)
5212 llvm::dbgs() << T->TokenText << " ";
5213 llvm::dbgs() << "\n";
5214 });
5215 } else {
5216 LLVM_DEBUG({
5217 llvm::dbgs() << "Did not expand macro \"" << ID->TokenText
5218 << "\", because it was used ";
5219 if (Args)
5220 llvm::dbgs() << "with " << Args->size();
5221 else
5222 llvm::dbgs() << "without";
5223 llvm::dbgs() << " arguments, which doesn't match any definition.\n";
5224 });
5225 Tokens->setPosition(Position);
5226 FormatTok = ID;
5227 PP = SavedPPState;
5228 }
5229 }
5230
5231 if (FormatTok->isNot(tok::comment)) {
5232 distributeComments(Comments, FormatTok);
5233 Comments.clear();
5234 return;
5235 }
5236
5237 Comments.push_back(FormatTok);
5238 } while (!eof());
5239
5240 distributeComments(Comments, nullptr);
5241 Comments.clear();
5242}
5243
5244namespace {
5245template <typename Iterator>
5246void pushTokens(Iterator Begin, Iterator End,
5248 for (auto I = Begin; I != End; ++I) {
5249 Into.push_back(I->Tok);
5250 for (const auto &Child : I->Children)
5251 pushTokens(Child.Tokens.begin(), Child.Tokens.end(), Into);
5252 }
5253}
5254} // namespace
5255
5256std::optional<llvm::SmallVector<llvm::SmallVector<FormatToken *, 8>, 1>>
5257UnwrappedLineParser::parseMacroCall(const PPState &SavedPPState) {
5258 std::optional<llvm::SmallVector<llvm::SmallVector<FormatToken *, 8>, 1>> Args;
5259 assert(Line->Tokens.empty());
5260 // Not nextToken(), which would already expand a directly following macro
5261 // call before the expansion of this one is inserted.
5262 auto ConsumeLastTokenOfCall = [this] {
5263 flushComments(isOnNewLine(*FormatTok));
5264 pushToken(FormatTok);
5265 FormatTok = Tokens->getNextToken();
5266 };
5267 if (Tokens->peekNextToken(/*SkipComment=*/true)->isNot(tok::l_paren)) {
5268 ConsumeLastTokenOfCall();
5269 return Args;
5270 }
5271 nextToken();
5272 assert(FormatTok->is(tok::l_paren));
5273 unsigned Position = Tokens->getPosition();
5274 FormatToken *Tok = FormatTok;
5275 nextToken();
5276 Args.emplace();
5277 auto ArgStart = std::prev(Line->Tokens.end());
5278
5279 int Parens = 0;
5280 do {
5281 switch (FormatTok->Tok.getKind()) {
5282 case tok::l_paren:
5283 ++Parens;
5284 nextToken();
5285 break;
5286 case tok::r_paren: {
5287 if (Parens > 0) {
5288 --Parens;
5289 nextToken();
5290 break;
5291 }
5292 Args->push_back({});
5293 pushTokens(std::next(ArgStart), Line->Tokens.end(), Args->back());
5294 ConsumeLastTokenOfCall();
5295 return Args;
5296 }
5297 case tok::comma: {
5298 if (Parens > 0) {
5299 nextToken();
5300 break;
5301 }
5302 Args->push_back({});
5303 pushTokens(std::next(ArgStart), Line->Tokens.end(), Args->back());
5304 nextToken();
5305 ArgStart = std::prev(Line->Tokens.end());
5306 break;
5307 }
5308 default:
5309 nextToken();
5310 break;
5311 }
5312 } while (!eof());
5313 Line->Tokens.resize(1);
5314 Tokens->setPosition(Position);
5315 FormatTok = Tok;
5316 PP = SavedPPState;
5317 return {};
5318}
5319
5320void UnwrappedLineParser::pushToken(FormatToken *Tok) {
5321 Line->Tokens.push_back(UnwrappedLineNode(Tok));
5322 if (PP.AtEndOfPPLine) {
5323 auto &Tok = *Line->Tokens.back().Tok;
5324 Tok.MustBreakBefore = true;
5325 Tok.MustBreakBeforeFinalized = true;
5326 Tok.FirstAfterPPLine = true;
5327 PP.AtEndOfPPLine = false;
5328 }
5329}
5330
5331} // end namespace format
5332} // end namespace clang
This file defines the FormatTokenSource interface, which provides a token stream as well as the abili...
This file contains the declaration of the FormatToken, a wrapper around Token with additional informa...
FormatToken()
Token Tok
The Token.
unsigned OriginalColumn
The original 0-based column of this token, including expanded tabs.
FormatToken * Previous
The previous token in the unwrapped line.
FormatToken * Next
The next token in the unwrapped line.
This file contains the main building blocks of macro support in clang-format.
static bool HasAttribute(const QualType &T)
This file implements a token annotator, i.e.
Defines the clang::TokenKind enum and support functions.
This file contains the declaration of the UnwrappedLineParser, which turns a stream of tokens into Un...
Implements an efficient mapping from strings to IdentifierInfo nodes.
Parser - This implements a parser for the C family of languages.
Definition Parser.h:266
This class handles loading and caching of source files into memory.
Token - This structure provides full information about a lexed token.
Definition Token.h:36
IdentifierInfo * getIdentifierInfo() const
Definition Token.h:197
bool isLiteral() const
Return true if this is a "literal", like a numeric constant, string, etc.
Definition Token.h:126
bool is(tok::TokenKind K) const
is/isNot - Predicates to check if this token is a specific kind, as in "if (Tok.is(tok::l_brace)) {....
Definition Token.h:104
tok::TokenKind getKind() const
Definition Token.h:99
bool isOneOf(Ts... Ks) const
Definition Token.h:105
bool isNot(tok::TokenKind K) const
Definition Token.h:111
CompoundStatementIndenter(UnwrappedLineParser *Parser, const FormatStyle &Style, unsigned &LineLevel)
CompoundStatementIndenter(UnwrappedLineParser *Parser, unsigned &LineLevel, bool WrapBrace, bool IndentBrace)
ScopedLineState(UnwrappedLineParser &Parser, bool SwitchToPreprocessorLines=false, bool DiscardLines=false)
Interface for users of the UnwrappedLineParser to receive the parsed lines.
UnwrappedLineParser(SourceManager &SourceMgr, const FormatStyle &Style, const AdditionalKeywords &Keywords, unsigned FirstStartColumn, ArrayRef< FormatToken * > Tokens, UnwrappedLineConsumer &Callback, llvm::SpecificBumpPtrAllocator< FormatToken > &Allocator, IdentifierTable &IdentTable)
static void hash_combine(std::size_t &seed, const T &v)
static bool mustBeJSIdentOrValue(const AdditionalKeywords &Keywords, const FormatToken *FormatTok)
std::ostream & operator<<(std::ostream &Stream, const UnwrappedLine &Line)
static bool tokenCanStartNewLine(const FormatToken &Tok)
static bool continuesLineCommentSection(const FormatToken &FormatTok, const UnwrappedLine &Line, const FormatStyle &Style, const llvm::Regex &CommentPragmasRegex)
static bool isC78Type(const FormatToken &Tok)
static bool isJSDeclOrStmt(const AdditionalKeywords &Keywords, const FormatToken *FormatTok)
LangOptions getFormattingLangOpts(const FormatStyle &Style=getLLVMStyle())
Returns the LangOpts that the formatter expects you to set.
Definition Format.cpp:4576
static void markOptionalBraces(FormatToken *LeftBrace)
static bool mustBeJSIdent(const AdditionalKeywords &Keywords, const FormatToken *FormatTok)
static bool isIIFE(const UnwrappedLine &Line, const AdditionalKeywords &Keywords)
static bool isC78ParameterDecl(const FormatToken *Tok, const FormatToken *Next, const FormatToken *FuncName)
static bool isGoogScope(const UnwrappedLine &Line)
static FormatToken * getLastNonComment(const UnwrappedLine &Line)
TokenType
Determines the semantic type of a syntactic token, e.g.
static bool ShouldBreakBeforeBrace(const FormatStyle &Style, const FormatToken &InitialToken, bool IsEmptyBlock, bool IsJavaRecord=false)
TokenKind
Provides a simple uniform namespace for tokens from all C languages.
Definition TokenKinds.h:33
bool isLiteral(TokenKind K)
Return true if this is a "literal" kind, like a numeric constant, string, etc.
Definition TokenKinds.h:109
Top level wrappers for InstallAPI frontend operations.
bool isLineComment(const FormatToken &FormatTok)
if(T->getSizeExpr()) TRY_TO(TraverseStmt(const_cast< Expr * >(T -> getSizeExpr())))
nullptr
This class represents a compute construct, representing a 'Kind' of ‘parallel’, 'serial',...
@ Default
Set to the current date and time.
const FunctionProtoType * T
bool continuesLineComment(const FormatToken &FormatTok, const FormatToken *Previous, const FormatToken *MinColumnToken)
@ Parens
New-expression has a C++98 paren-delimited initializer.
Definition ExprCXX.h:2250
@ LK_C
Should be used for C.
Definition Format.h:3863
@ LK_Proto
Should be used for Protocol Buffers
Definition Format.h:3877
@ IEBS_AfterExternBlock
Backwards compatible with AfterExternBlock's indenting.
Definition Format.h:3301
@ IEBS_Indent
Indents extern blocks.
Definition Format.h:3315
@ PPDIS_BeforeHash
Indents directives before the hash.
Definition Format.h:3408
@ PPDIS_None
Does not indent any directives.
Definition Format.h:3390
@ LS_Cpp20
Parse and format as C++20.
Definition Format.h:5923
@ BWACS_Always
Always wrap braces after a control statement.
Definition Format.h:1415
@ BWACS_Never
Never wrap braces after a control statement.
Definition Format.h:1394
@ IAMS_Never
Use AccessModifierOffset for access modifiers and indent members one level below the record.
Definition Format.h:3196
@ IAMS_AfterFirstAccessModifier
In C, C++, and Objective-C, indent members one level until the first explicit access modifier,...
Definition Format.h:3220
@ IAMS_Always
Give access modifiers their own indentation level and indent all members two levels below the record.
Definition Format.h:3208
@ BS_Whitesmiths
Like Allman but always indent braces and line up code with braces.
Definition Format.h:2250
@ RPS_Leave
Do not remove parentheses.
Definition Format.h:4848
@ RPS_ReturnStatement
Also remove parentheses enclosing the expression in a return/co_return statement.
Definition Format.h:4863
@ NI_All
Indent in all namespaces.
Definition Format.h:4045
@ NI_Inner
Indent only in inner namespaces (nested in other namespaces).
Definition Format.h:4035
@ IGLS_OuterIndent
Indent goto labels to the enclosing block (previous indenting level).
Definition Format.h:3347
@ IGLS_NoIndent
Do not indent goto labels.
Definition Format.h:3335
Encapsulates keywords that are context sensitive or for languages not properly supported by Clang's l...
IdentifierInfo * kw_instanceof
IdentifierInfo * kw_implements
IdentifierInfo * kw_override
IdentifierInfo * kw_await
IdentifierInfo * kw_extends
IdentifierInfo * kw_async
IdentifierInfo * kw_from
IdentifierInfo * kw_abstract
IdentifierInfo * kw_var
IdentifierInfo * kw_interface
IdentifierInfo * kw_function
IdentifierInfo * kw_yield
IdentifierInfo * kw_where
IdentifierInfo * kw_throws
IdentifierInfo * kw_let
IdentifierInfo * kw_import
IdentifierInfo * kw_finally
Represents a complete lambda introducer.
Definition DeclSpec.h:2887
The FormatStyle is used to configure the formatting to follow specific guidelines.
Definition Format.h:51
@ LK_Proto
Should be used for Protocol Buffers
Definition Format.h:3877
@ RCS_Always
Apply indentation rules and reflow long comments into new lines, trying to obey the ColumnLimit.
Definition Format.h:4755
@ SRS_Empty
Only merge empty records.
Definition Format.h:1083
@ BS_Whitesmiths
Like Allman but always indent braces and line up code with braces.
Definition Format.h:2250
A wrapper around a Token storing information about the whitespace characters preceding it.
bool Optional
Is optional and can be removed.
bool isNot(T Kind) const
StringRef TokenText
The raw text of the token.
bool isNoneOf(Ts... Ks) const
unsigned NewlinesBefore
The number of newlines immediately before the Token.
bool is(tok::TokenKind Kind) const
bool isOneOf(A K1, B K2) const
unsigned IsFirst
Indicates that this is the first token of the file.
FormatToken * MatchingParen
If this is a bracket, this points to the matching one.
FormatToken * Previous
The previous token in the unwrapped line.
An unwrapped line is a sequence of Token, that we would like to put on a single line if there was no ...