clang 24.0.0git
UnwrappedLineParser.cpp
Go to the documentation of this file.
1//===--- UnwrappedLineParser.cpp - Format C++ code ------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8///
9/// \file
10/// This file contains the implementation of the UnwrappedLineParser,
11/// which turns a stream of tokens into UnwrappedLines.
12///
13//===----------------------------------------------------------------------===//
14
15#include "UnwrappedLineParser.h"
16#include "FormatToken.h"
17#include "FormatTokenSource.h"
18#include "Macros.h"
19#include "TokenAnnotator.h"
21#include "llvm/ADT/STLExtras.h"
22#include "llvm/ADT/StringRef.h"
23#include "llvm/Support/Debug.h"
24#include "llvm/Support/raw_os_ostream.h"
25#include "llvm/Support/raw_ostream.h"
26
27#include <utility>
28
29#define DEBUG_TYPE "format-parser"
30
31namespace clang {
32namespace format {
33
34namespace {
35
36void printLine(llvm::raw_ostream &OS, const UnwrappedLine &Line,
37 StringRef Prefix = "", bool PrintText = false) {
38 OS << Prefix << "Line(" << Line.Level << ", FSC=" << Line.FirstStartColumn
39 << ")" << (Line.InPPDirective ? " MACRO" : "") << ": ";
40 bool NewLine = false;
41 for (std::list<UnwrappedLineNode>::const_iterator I = Line.Tokens.begin(),
42 E = Line.Tokens.end();
43 I != E; ++I) {
44 if (NewLine) {
45 OS << Prefix;
46 NewLine = false;
47 }
48 OS << I->Tok->Tok.getName() << "["
49 << "T=" << (unsigned)I->Tok->getType()
50 << ", OC=" << I->Tok->OriginalColumn << ", \"" << I->Tok->TokenText
51 << "\"] ";
52 for (const auto *CI = I->Children.begin(), *CE = I->Children.end();
53 CI != CE; ++CI) {
54 OS << "\n";
55 printLine(OS, *CI, (Prefix + " ").str());
56 NewLine = true;
57 }
58 }
59 if (!NewLine)
60 OS << "\n";
61}
62
63[[maybe_unused]] static void printDebugInfo(const UnwrappedLine &Line) {
64 printLine(llvm::dbgs(), Line);
65}
66
67class ScopedDeclarationState {
68public:
69 ScopedDeclarationState(UnwrappedLine &Line, llvm::BitVector &Stack,
70 bool MustBeDeclaration)
71 : Line(Line), Stack(Stack) {
72 Line.MustBeDeclaration = MustBeDeclaration;
73 Stack.push_back(MustBeDeclaration);
74 }
75 ~ScopedDeclarationState() {
76 Stack.pop_back();
77 if (!Stack.empty())
78 Line.MustBeDeclaration = Stack.back();
79 else
80 Line.MustBeDeclaration = true;
81 }
82
83private:
84 UnwrappedLine &Line;
85 llvm::BitVector &Stack;
86};
87
88} // end anonymous namespace
89
90std::ostream &operator<<(std::ostream &Stream, const UnwrappedLine &Line) {
91 llvm::raw_os_ostream OS(Stream);
92 printLine(OS, Line);
93 return Stream;
94}
95
97public:
98 // With \c DiscardLines, the lines added while in scope are discarded.
100 bool SwitchToPreprocessorLines = false,
101 bool DiscardLines = false)
102 : Parser(Parser), OriginalLines(Parser.CurrentLines),
103 DiscardLines(DiscardLines) {
104 if (SwitchToPreprocessorLines)
105 Parser.CurrentLines = &Parser.PreprocessorDirectives;
106 else if (!Parser.Line->Tokens.empty())
107 Parser.CurrentLines = &Parser.Line->Tokens.back().Children;
108 OriginalNumLines = Parser.CurrentLines->size();
109 PreBlockLine = std::move(Parser.Line);
110 Parser.Line = std::make_unique<UnwrappedLine>();
111 Parser.Line->Level = PreBlockLine->Level;
112 Parser.Line->PPLevel = PreBlockLine->PPLevel;
113 Parser.Line->InPPDirective = PreBlockLine->InPPDirective;
114 Parser.Line->InMacroBody = PreBlockLine->InMacroBody;
115 Parser.Line->UnbracedBodyLevel = PreBlockLine->UnbracedBodyLevel;
116 }
117
119 if (!Parser.Line->Tokens.empty())
120 Parser.addUnwrappedLine();
121 assert(Parser.Line->Tokens.empty());
122 if (DiscardLines)
123 Parser.CurrentLines->truncate(OriginalNumLines);
124 Parser.Line = std::move(PreBlockLine);
125 if (Parser.CurrentLines == &Parser.PreprocessorDirectives)
126 Parser.PP.AtEndOfPPLine = true;
127 Parser.CurrentLines = OriginalLines;
128 }
129
130private:
132
133 std::unique_ptr<UnwrappedLine> PreBlockLine;
134 SmallVectorImpl<UnwrappedLine> *OriginalLines;
135 size_t OriginalNumLines;
136 bool DiscardLines;
137};
138
140public:
142 const FormatStyle &Style, unsigned &LineLevel)
144 Style.BraceWrapping.AfterControlStatement ==
145 FormatStyle::BWACS_Always,
146 Style.BraceWrapping.IndentBraces) {}
148 bool WrapBrace, bool IndentBrace)
149 : LineLevel(LineLevel), OldLineLevel(LineLevel) {
150 if (WrapBrace)
151 Parser->addUnwrappedLine();
152 if (IndentBrace)
153 ++LineLevel;
154 }
155 ~CompoundStatementIndenter() { LineLevel = OldLineLevel; }
156
157private:
158 unsigned &LineLevel;
159 unsigned OldLineLevel;
160};
161
163 SourceManager &SourceMgr, const FormatStyle &Style,
164 const AdditionalKeywords &Keywords, unsigned FirstStartColumn,
166 llvm::SpecificBumpPtrAllocator<FormatToken> &Allocator,
167 IdentifierTable &IdentTable)
168 : Line(new UnwrappedLine), CurrentLines(&Lines), Style(Style),
169 IsCpp(Style.isCpp()), LangOpts(getFormattingLangOpts(Style)),
170 Keywords(Keywords), CommentPragmasRegex(Style.CommentPragmas),
171 Tokens(nullptr), Callback(Callback), AllTokens(Tokens),
172 PP(getIncludeGuardState(Style.IndentPPDirectives)),
173 FirstStartColumn(FirstStartColumn),
174 Macros(Style.Macros, SourceMgr, Style, Allocator, IdentTable) {}
175
176void UnwrappedLineParser::reset() {
177 PP.BranchLevel = -1;
178 PP.IncludeGuard = getIncludeGuardState(Style.IndentPPDirectives);
179 PP.IncludeGuardToken = nullptr;
180 ParsedPPDirectives.clear();
181 Line.reset(new UnwrappedLine);
182 CommentsBeforeNextToken.clear();
183 FormatTok = nullptr;
184 PP.AtEndOfPPLine = false;
185 IsDecltypeAutoFunction = false;
186 PreprocessorDirectives.clear();
187 CurrentLines = &Lines;
188 DeclarationScopeStack.clear();
189 NestedTooDeep.clear();
190 NestedLambdas.clear();
191 PP.Stack.clear();
192 Line->FirstStartColumn = FirstStartColumn;
193
194 if (!Unexpanded.empty())
195 for (FormatToken *Token : AllTokens)
196 Token->MacroCtx.reset();
197 CurrentExpandedLines.clear();
198 ExpandedLines.clear();
199 Unexpanded.clear();
200 InExpansion = false;
201 Reconstruct.reset();
202}
203
205 IndexedTokenSource TokenSource(AllTokens);
206 Line->FirstStartColumn = FirstStartColumn;
207 do {
208 LLVM_DEBUG(llvm::dbgs() << "----\n");
209 reset();
210 Tokens = &TokenSource;
211 TokenSource.reset();
212
213 readToken();
214 parseFile();
215
216 // If we found an include guard then all preprocessor directives (other than
217 // the guard) are over-indented by one.
218 if (PP.IncludeGuard == IG_Found) {
219 for (auto &Line : Lines)
220 if (Line.InPPDirective && Line.Level > 0)
221 --Line.Level;
222 }
223
224 // Create line with eof token.
225 assert(eof());
226 pushToken(FormatTok);
227 addUnwrappedLine();
228
229 // In a first run, format everything with the lines containing macro calls
230 // replaced by the expansion.
231 if (!ExpandedLines.empty()) {
232 LLVM_DEBUG(llvm::dbgs() << "Expanded lines:\n");
233 for (const auto &Line : Lines) {
234 if (!Line.Tokens.empty()) {
235 auto it = ExpandedLines.find(Line.Tokens.begin()->Tok);
236 if (it != ExpandedLines.end()) {
237 for (const auto &Expanded : it->second) {
238 LLVM_DEBUG(printDebugInfo(Expanded));
239 Callback.consumeUnwrappedLine(Expanded);
240 }
241 continue;
242 }
243 }
244 LLVM_DEBUG(printDebugInfo(Line));
245 Callback.consumeUnwrappedLine(Line);
246 }
247 Callback.finishRun();
248 }
249
250 LLVM_DEBUG(llvm::dbgs() << "Unwrapped lines:\n");
251 for (const UnwrappedLine &Line : Lines) {
252 LLVM_DEBUG(printDebugInfo(Line));
253 Callback.consumeUnwrappedLine(Line);
254 }
255 Callback.finishRun();
256 Lines.clear();
257 while (!PP.LevelBranchIndex.empty() &&
258 PP.LevelBranchIndex.back() + 1 >= PP.LevelBranchCount.back()) {
259 PP.LevelBranchIndex.resize(PP.LevelBranchIndex.size() - 1);
260 PP.LevelBranchCount.resize(PP.LevelBranchCount.size() - 1);
261 }
262 if (!PP.LevelBranchIndex.empty()) {
263 ++PP.LevelBranchIndex.back();
264 assert(PP.LevelBranchIndex.size() == PP.LevelBranchCount.size());
265 assert(PP.LevelBranchIndex.back() <= PP.LevelBranchCount.back());
266 }
267 } while (!PP.LevelBranchIndex.empty());
268}
269
270void UnwrappedLineParser::parseFile() {
271 // The top-level context in a file always has declarations, except for pre-
272 // processor directives and JavaScript files.
273 bool MustBeDeclaration = !Line->InPPDirective && !Style.isJavaScript();
274 ScopedDeclarationState DeclarationState(*Line, DeclarationScopeStack,
275 MustBeDeclaration);
276 if (Style.isTextProto() || (Style.isJson() && FormatTok->IsFirst))
277 parseBracedList();
278 else
279 parseLevel();
280 // Make sure to format the remaining tokens.
281 //
282 // LK_TextProto is special since its top-level is parsed as the body of a
283 // braced list, which does not necessarily have natural line separators such
284 // as a semicolon. Comments after the last entry that have been determined to
285 // not belong to that line, as in:
286 // key: value
287 // // endfile comment
288 // do not have a chance to be put on a line of their own until this point.
289 // Here we add this newline before end-of-file comments.
290 if (Style.isTextProto() && !CommentsBeforeNextToken.empty())
291 addUnwrappedLine();
292 flushComments(true);
293 addUnwrappedLine();
294}
295
296void UnwrappedLineParser::parseCSharpGenericTypeConstraint() {
297 do {
298 switch (FormatTok->Tok.getKind()) {
299 case tok::l_brace:
300 case tok::semi:
301 return;
302 default:
303 if (FormatTok->is(Keywords.kw_where)) {
304 addUnwrappedLine();
305 nextToken();
306 parseCSharpGenericTypeConstraint();
307 break;
308 }
309 nextToken();
310 break;
311 }
312 } while (!eof());
313}
314
315void UnwrappedLineParser::parseCSharpAttribute() {
316 int UnpairedSquareBrackets = 1;
317 do {
318 switch (FormatTok->Tok.getKind()) {
319 case tok::r_square:
320 nextToken();
321 --UnpairedSquareBrackets;
322 if (UnpairedSquareBrackets == 0) {
323 addUnwrappedLine();
324 return;
325 }
326 break;
327 case tok::l_square:
328 ++UnpairedSquareBrackets;
329 nextToken();
330 break;
331 default:
332 nextToken();
333 break;
334 }
335 } while (!eof());
336}
337
338bool UnwrappedLineParser::precededByCommentOrPPDirective() const {
339 if (!Lines.empty() && Lines.back().InPPDirective)
340 return true;
341
342 const FormatToken *Previous = Tokens->getPreviousToken();
343 return Previous && Previous->is(tok::comment) &&
344 (Previous->IsMultiline || Previous->NewlinesBefore > 0);
345}
346
347/// Parses a level, that is ???.
348/// \param OpeningBrace Opening brace (\p nullptr if absent) of that level.
349/// \param IfKind The \p if statement kind in the level.
350/// \param IfLeftBrace The left brace of the \p if block in the level.
351/// \returns true if a simple block of if/else/for/while, or false otherwise.
352/// (A simple block has a single statement.)
353bool UnwrappedLineParser::parseLevel(const FormatToken *OpeningBrace,
354 IfStmtKind *IfKind,
355 FormatToken **IfLeftBrace) {
356 const bool InRequiresExpression =
357 OpeningBrace && OpeningBrace->is(TT_RequiresExpressionLBrace);
358 const bool IsPrecededByCommentOrPPDirective =
359 !Style.RemoveBracesLLVM || precededByCommentOrPPDirective();
360 FormatToken *IfLBrace = nullptr;
361 bool HasDoWhile = false;
362 bool HasLabel = false;
363 unsigned StatementCount = 0;
364 bool SwitchLabelEncountered = false;
365
366 do {
367 if (FormatTok->isAttribute()) {
368 nextToken();
369 if (FormatTok->is(tok::l_paren))
370 parseParens();
371 continue;
372 }
373 tok::TokenKind Kind = FormatTok->Tok.getKind();
374 if (FormatTok->is(TT_MacroBlockBegin))
375 Kind = tok::l_brace;
376 else if (FormatTok->is(TT_MacroBlockEnd))
377 Kind = tok::r_brace;
378
379 auto ParseDefault = [this, OpeningBrace, IfKind, &IfLBrace, &HasDoWhile,
380 &HasLabel, &StatementCount] {
381 parseStructuralElement(OpeningBrace, IfKind, &IfLBrace,
382 HasDoWhile ? nullptr : &HasDoWhile,
383 HasLabel ? nullptr : &HasLabel);
384 ++StatementCount;
385 assert(StatementCount > 0 && "StatementCount overflow!");
386 };
387
388 switch (Kind) {
389 case tok::comment:
390 nextToken();
391 addUnwrappedLine();
392 break;
393 case tok::l_brace:
394 if (InRequiresExpression) {
395 FormatTok->setFinalizedType(TT_CompoundRequirementLBrace);
396 } else if (FormatTok->Previous &&
397 FormatTok->Previous->ClosesRequiresClause) {
398 // We need the 'default' case here to correctly parse a function
399 // l_brace.
400 ParseDefault();
401 continue;
402 }
403 if (!InRequiresExpression && FormatTok->isNot(TT_MacroBlockBegin)) {
404 if (tryToParseBracedList())
405 continue;
406 FormatTok->setFinalizedType(TT_BlockLBrace);
407 }
408 parseBlock();
409 ++StatementCount;
410 assert(StatementCount > 0 && "StatementCount overflow!");
411 addUnwrappedLine();
412 break;
413 case tok::r_brace:
414 if (OpeningBrace) {
415 if (!Style.RemoveBracesLLVM || Line->InPPDirective ||
416 OpeningBrace->isNoneOf(TT_ControlStatementLBrace, TT_ElseLBrace)) {
417 return false;
418 }
419 if (FormatTok->isNot(tok::r_brace) || StatementCount != 1 || HasLabel ||
420 HasDoWhile || IsPrecededByCommentOrPPDirective ||
421 precededByCommentOrPPDirective()) {
422 return false;
423 }
424 const FormatToken *Next = Tokens->peekNextToken();
425 if (Next->is(tok::comment) && Next->NewlinesBefore == 0)
426 return false;
427 if (IfLeftBrace)
428 *IfLeftBrace = IfLBrace;
429 return true;
430 }
431 nextToken();
432 addUnwrappedLine();
433 break;
434 case tok::kw_default: {
435 unsigned StoredPosition = Tokens->getPosition();
436 auto *Next = Tokens->getNextNonComment();
437 FormatTok = Tokens->setPosition(StoredPosition);
438 if (Next->isNoneOf(tok::colon, tok::arrow)) {
439 // default not followed by `:` or `->` is not a case label; treat it
440 // like an identifier.
441 parseStructuralElement();
442 break;
443 }
444 // Else, if it is 'default:', fall through to the case handling.
445 [[fallthrough]];
446 }
447 case tok::kw_case:
448 if (Style.Language == FormatStyle::LK_Proto || Style.isVerilog() ||
449 (Style.isJavaScript() && Line->MustBeDeclaration)) {
450 // Proto: there are no switch/case statements
451 // Verilog: Case labels don't have this word. We handle case
452 // labels including default in TokenAnnotator.
453 // JavaScript: A 'case: string' style field declaration.
454 ParseDefault();
455 break;
456 }
457 if (!SwitchLabelEncountered &&
458 (Style.IndentCaseLabels ||
459 (OpeningBrace && OpeningBrace->is(TT_SwitchExpressionLBrace)) ||
460 (Line->InPPDirective && Line->Level == 1))) {
461 ++Line->Level;
462 }
463 SwitchLabelEncountered = true;
464 parseStructuralElement();
465 break;
466 case tok::l_square:
467 if (Style.isCSharp()) {
468 nextToken();
469 parseCSharpAttribute();
470 break;
471 }
472 if (handleCppAttributes())
473 break;
474 [[fallthrough]];
475 default:
476 ParseDefault();
477 break;
478 }
479 } while (!eof());
480
481 return false;
482}
483
484void UnwrappedLineParser::calculateBraceTypes(bool ExpectClassBody) {
485 // We'll parse forward through the tokens until we hit
486 // a closing brace or eof - note that getNextToken() will
487 // parse macros, so this will magically work inside macro
488 // definitions, too.
489 unsigned StoredPosition = Tokens->getPosition();
490 FormatToken *Tok = FormatTok;
491 const FormatToken *PrevTok = Tok->Previous;
492 // Keep a stack of positions of lbrace tokens. We will
493 // update information about whether an lbrace starts a
494 // braced init list or a different block during the loop.
495 struct StackEntry {
497 const FormatToken *PrevTok;
498 };
499 SmallVector<StackEntry, 8> LBraceStack;
500 assert(Tok->is(tok::l_brace));
501
502 do {
503 auto *NextTok = Tokens->getNextNonComment();
504
505 if (!Line->InMacroBody && !Style.isTableGen()) {
506 // Skip PPDirective lines (except macro definitions) and comments.
507 while (NextTok->is(tok::hash)) {
508 NextTok = Tokens->getNextToken();
509 if (NextTok->isOneOf(tok::pp_not_keyword, tok::pp_define))
510 break;
511 do {
512 NextTok = Tokens->getNextToken();
513 } while (!NextTok->HasUnescapedNewline && NextTok->isNot(tok::eof));
514
515 while (NextTok->is(tok::comment))
516 NextTok = Tokens->getNextToken();
517 }
518 }
519
520 switch (Tok->Tok.getKind()) {
521 case tok::l_brace:
522 if (Style.isJavaScript() && PrevTok) {
523 if (PrevTok->isOneOf(tok::colon, tok::less)) {
524 // A ':' indicates this code is in a type, or a braced list
525 // following a label in an object literal ({a: {b: 1}}).
526 // A '<' could be an object used in a comparison, but that is nonsense
527 // code (can never return true), so more likely it is a generic type
528 // argument (`X<{a: string; b: number}>`).
529 // The code below could be confused by semicolons between the
530 // individual members in a type member list, which would normally
531 // trigger BK_Block. In both cases, this must be parsed as an inline
532 // braced init.
533 Tok->setBlockKind(BK_BracedInit);
534 } else if (PrevTok->is(tok::r_paren)) {
535 // `) { }` can only occur in function or method declarations in JS.
536 Tok->setBlockKind(BK_Block);
537 }
538 } else if (Style.isJava() && PrevTok && PrevTok->is(tok::arrow)) {
539 Tok->setBlockKind(BK_Block);
540 } else {
541 Tok->setBlockKind(BK_Unknown);
542 }
543 LBraceStack.push_back({Tok, PrevTok});
544 break;
545 case tok::r_brace:
546 if (LBraceStack.empty())
547 break;
548 if (auto *LBrace = LBraceStack.back().Tok; LBrace->is(BK_Unknown)) {
549 bool ProbablyBracedList = false;
550 if (Style.Language == FormatStyle::LK_Proto) {
551 ProbablyBracedList = NextTok->isOneOf(tok::comma, tok::r_square);
552 } else if (LBrace->isNot(TT_EnumLBrace)) {
553 // Using OriginalColumn to distinguish between ObjC methods and
554 // binary operators is a bit hacky.
555 bool NextIsObjCMethod = NextTok->isOneOf(tok::plus, tok::minus) &&
556 NextTok->OriginalColumn == 0;
557
558 // Try to detect a braced list. Note that regardless how we mark inner
559 // braces here, we will overwrite the BlockKind later if we parse a
560 // braced list (where all blocks inside are by default braced lists),
561 // or when we explicitly detect blocks (for example while parsing
562 // lambdas).
563
564 // If we already marked the opening brace as braced list, the closing
565 // must also be part of it.
566 ProbablyBracedList = LBrace->is(TT_BracedListLBrace);
567
568 ProbablyBracedList = ProbablyBracedList ||
569 (Style.isJavaScript() &&
570 NextTok->isOneOf(Keywords.kw_of, Keywords.kw_in,
571 Keywords.kw_as));
572 ProbablyBracedList =
573 ProbablyBracedList ||
574 (IsCpp && (PrevTok->Tok.isLiteral() ||
575 NextTok->isOneOf(tok::l_paren, tok::arrow)));
576
577 // If there is a comma, or right paren after the closing brace, we
578 // assume this is a braced initializer list.
579 // FIXME: Some of these do not apply to JS, e.g. "} {" can never be a
580 // braced list in JS.
581 ProbablyBracedList =
582 ProbablyBracedList ||
583 NextTok->isOneOf(tok::comma, tok::period, tok::colon,
584 tok::r_paren, tok::r_square, tok::ellipsis);
585
586 // Distinguish between braced list in a constructor initializer list
587 // followed by constructor body, or just adjacent blocks.
588 ProbablyBracedList =
589 ProbablyBracedList ||
590 (NextTok->is(tok::l_brace) && LBraceStack.back().PrevTok &&
591 LBraceStack.back().PrevTok->isOneOf(tok::identifier,
592 tok::greater));
593
594 ProbablyBracedList =
595 ProbablyBracedList ||
596 (NextTok->is(tok::identifier) &&
597 PrevTok->isNoneOf(tok::semi, tok::r_brace, tok::l_brace));
598
599 ProbablyBracedList = ProbablyBracedList ||
600 (NextTok->is(tok::semi) &&
601 (!ExpectClassBody || LBraceStack.size() != 1));
602
603 ProbablyBracedList =
604 ProbablyBracedList ||
605 (NextTok->isBinaryOperator() && !NextIsObjCMethod);
606
607 if (!Style.isCSharp() && NextTok->is(tok::l_square)) {
608 // We can have an array subscript after a braced init
609 // list, but C++11 attributes are expected after blocks.
610 NextTok = Tokens->getNextToken();
611 ProbablyBracedList = NextTok->isNot(tok::l_square);
612 }
613
614 // Cpp macro definition body that is a nonempty braced list or block:
615 if (IsCpp && Line->InMacroBody && PrevTok != FormatTok &&
616 !FormatTok->Previous && NextTok->is(tok::eof) &&
617 // A statement can end with only `;` (simple statement), a block
618 // closing brace (compound statement), or `:` (label statement).
619 // If PrevTok is a block opening brace, Tok ends an empty block.
620 PrevTok->isNoneOf(tok::semi, BK_Block, tok::colon)) {
621 ProbablyBracedList = true;
622 }
623 }
624 const auto BlockKind = ProbablyBracedList ? BK_BracedInit : BK_Block;
625 Tok->setBlockKind(BlockKind);
626 LBrace->setBlockKind(BlockKind);
627 }
628 LBraceStack.pop_back();
629 break;
630 case tok::identifier:
631 if (Tok->isNot(TT_StatementMacro))
632 break;
633 [[fallthrough]];
634 case tok::at:
635 case tok::semi:
636 case tok::kw_if:
637 case tok::kw_while:
638 case tok::kw_for:
639 case tok::kw_switch:
640 case tok::kw_try:
641 case tok::kw___try:
642 if (!LBraceStack.empty() && LBraceStack.back().Tok->is(BK_Unknown))
643 LBraceStack.back().Tok->setBlockKind(BK_Block);
644 break;
645 default:
646 break;
647 }
648
649 PrevTok = Tok;
650 Tok = NextTok;
651 } while (Tok->isNot(tok::eof) && !LBraceStack.empty());
652
653 // Assume other blocks for all unclosed opening braces.
654 for (const auto &Entry : LBraceStack)
655 if (Entry.Tok->is(BK_Unknown))
656 Entry.Tok->setBlockKind(BK_Block);
657
658 FormatTok = Tokens->setPosition(StoredPosition);
659}
660
661// Sets the token type of the directly previous right brace.
662void UnwrappedLineParser::setPreviousRBraceType(TokenType Type) {
663 if (auto Prev = FormatTok->getPreviousNonComment();
664 Prev && Prev->is(tok::r_brace)) {
665 Prev->setFinalizedType(Type);
666 }
667}
668
669template <class T>
670static inline void hash_combine(std::size_t &seed, const T &v) {
671 std::hash<T> hasher;
672 seed ^= hasher(v) + 0x9e3779b9 + (seed << 6) + (seed >> 2);
673}
674
675size_t UnwrappedLineParser::computePPHash() const {
676 size_t h = 0;
677 for (const auto &i : PP.Stack) {
678 hash_combine(h, size_t(i.Kind));
679 hash_combine(h, i.Line);
680 }
681 return h;
682}
683
684// Checks whether \p ParsedLine might fit on a single line. If \p OpeningBrace
685// is not null, subtracts its length (plus the preceding space) when computing
686// the length of \p ParsedLine. We must clone the tokens of \p ParsedLine before
687// running the token annotator on it so that we can restore them afterward.
688bool UnwrappedLineParser::mightFitOnOneLine(
689 UnwrappedLine &ParsedLine, const FormatToken *OpeningBrace) const {
690 const auto ColumnLimit = Style.ColumnLimit;
691 if (ColumnLimit == 0)
692 return true;
693
694 auto &Tokens = ParsedLine.Tokens;
695 assert(!Tokens.empty());
696
697 const auto *LastToken = Tokens.back().Tok;
698 assert(LastToken);
699
700 SmallVector<UnwrappedLineNode> SavedTokens(Tokens.size());
701
702 int Index = 0;
703 for (const auto &Token : Tokens) {
704 assert(Token.Tok);
705 auto &SavedToken = SavedTokens[Index++];
706 SavedToken.Tok = new FormatToken;
707 SavedToken.Tok->copyFrom(*Token.Tok);
708 SavedToken.Children = std::move(Token.Children);
709 }
710
711 AnnotatedLine Line(ParsedLine);
712 assert(Line.Last == LastToken);
713
714 TokenAnnotator Annotator(Style, Keywords);
715 Annotator.annotate(Line);
716 Annotator.calculateFormattingInformation(Line);
717
718 auto Length = LastToken->TotalLength;
719 if (OpeningBrace) {
720 assert(OpeningBrace != Tokens.front().Tok);
721 if (auto Prev = OpeningBrace->Previous;
722 Prev && Prev->TotalLength + ColumnLimit == OpeningBrace->TotalLength) {
723 Length -= ColumnLimit;
724 }
725 Length -= OpeningBrace->TokenText.size() + 1;
726 }
727
728 if (const auto *FirstToken = Line.First; FirstToken->is(tok::r_brace)) {
729 assert(!OpeningBrace || OpeningBrace->is(TT_ControlStatementLBrace));
730 Length -= FirstToken->TokenText.size() + 1;
731 }
732
733 Index = 0;
734 for (auto &Token : Tokens) {
735 const auto &SavedToken = SavedTokens[Index++];
736 Token.Tok->copyFrom(*SavedToken.Tok);
737 Token.Children = std::move(SavedToken.Children);
738 delete SavedToken.Tok;
739 }
740
741 // If these change PPLevel needs to be used for get correct indentation.
742 assert(!Line.InMacroBody);
743 assert(!Line.InPPDirective);
744 return Line.Level * Style.IndentWidth + Length <= ColumnLimit;
745}
746
747FormatToken *UnwrappedLineParser::parseBlock(bool MustBeDeclaration,
748 unsigned AddLevels, bool MunchSemi,
749 bool KeepBraces,
750 IfStmtKind *IfKind,
751 bool UnindentWhitesmithsBraces) {
752 auto HandleVerilogBlockLabel = [this]() {
753 // ":" name
754 if (Style.isVerilog() && FormatTok->is(tok::colon)) {
755 nextToken();
756 if (Keywords.isVerilogIdentifier(*FormatTok))
757 nextToken();
758 }
759 };
760
761 // Whether this is a Verilog-specific block that has a special header like a
762 // module.
763 const bool VerilogHierarchy =
764 Style.isVerilog() && Keywords.isVerilogHierarchy(*FormatTok);
765 assert((FormatTok->isOneOf(tok::l_brace, TT_MacroBlockBegin) ||
766 (Style.isVerilog() &&
767 (Keywords.isVerilogBegin(*FormatTok) || VerilogHierarchy))) &&
768 "'{' or macro block token expected");
769 FormatToken *Tok = FormatTok;
770 const bool FollowedByComment = Tokens->peekNextToken()->is(tok::comment);
771 auto Index = CurrentLines->size();
772 const bool MacroBlock = FormatTok->is(TT_MacroBlockBegin);
773 FormatTok->setBlockKind(BK_Block);
774
775 const bool IsWhitesmiths =
776 Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths;
777
778 // For Whitesmiths mode, jump to the next level prior to skipping over the
779 // braces.
780 if (!VerilogHierarchy && AddLevels > 0 && IsWhitesmiths)
781 ++Line->Level;
782
783 size_t PPStartHash = computePPHash();
784
785 const unsigned InitialLevel = Line->Level;
786 if (VerilogHierarchy) {
787 AddLevels += parseVerilogHierarchyHeader();
788 } else {
789 nextToken(/*LevelDifference=*/AddLevels);
790 HandleVerilogBlockLabel();
791 }
792
793 // Bail out if there are too many levels. Otherwise, the stack might overflow.
794 if (Line->Level > 300)
795 return nullptr;
796
797 if (MacroBlock && FormatTok->is(tok::l_paren))
798 parseParens();
799
800 size_t NbPreprocessorDirectives =
801 !parsingPPDirective() ? PreprocessorDirectives.size() : 0;
802 addUnwrappedLine();
803 size_t OpeningLineIndex =
804 CurrentLines->empty()
806 : (CurrentLines->size() - 1 - NbPreprocessorDirectives);
807
808 // Whitesmiths is weird here. The brace needs to be indented for the namespace
809 // block, but the block itself may not be indented depending on the style
810 // settings. This allows the format to back up one level in those cases.
811 if (UnindentWhitesmithsBraces)
812 --Line->Level;
813
814 ScopedDeclarationState DeclarationState(*Line, DeclarationScopeStack,
815 MustBeDeclaration);
816
817 // Whitesmiths logic has already added a level by this point, so avoid
818 // adding it twice.
819 if (AddLevels > 0u)
820 Line->Level += AddLevels - (IsWhitesmiths ? 1 : 0);
821
822 FormatToken *IfLBrace = nullptr;
823 const bool SimpleBlock = parseLevel(Tok, IfKind, &IfLBrace);
824
825 if (eof())
826 return IfLBrace;
827
828 if (MacroBlock ? FormatTok->isNot(TT_MacroBlockEnd)
829 : FormatTok->isNot(tok::r_brace)) {
830 Line->Level = InitialLevel;
831 FormatTok->setBlockKind(BK_Block);
832 return IfLBrace;
833 }
834
835 if (FormatTok->is(tok::r_brace)) {
836 FormatTok->setBlockKind(BK_Block);
837 if (Tok->is(TT_NamespaceLBrace))
838 FormatTok->setFinalizedType(TT_NamespaceRBrace);
839 }
840
841 const bool IsFunctionRBrace =
842 FormatTok->is(tok::r_brace) && Tok->is(TT_FunctionLBrace);
843
844 auto RemoveBraces = [=]() mutable {
845 if (!SimpleBlock)
846 return false;
847 assert(Tok->isOneOf(TT_ControlStatementLBrace, TT_ElseLBrace));
848 assert(FormatTok->is(tok::r_brace));
849 const bool WrappedOpeningBrace = !Tok->Previous;
850 if (WrappedOpeningBrace && FollowedByComment)
851 return false;
852 const bool HasRequiredIfBraces = IfLBrace && !IfLBrace->Optional;
853 if (KeepBraces && !HasRequiredIfBraces)
854 return false;
855 if (Tok->isNot(TT_ElseLBrace) || !HasRequiredIfBraces) {
856 const FormatToken *Previous = Tokens->getPreviousToken();
857 assert(Previous);
858 if (Previous->is(tok::r_brace) && !Previous->Optional)
859 return false;
860 }
861 assert(!CurrentLines->empty());
862 auto &LastLine = CurrentLines->back();
863 if (LastLine.Level == InitialLevel + 1 && !mightFitOnOneLine(LastLine))
864 return false;
865 if (Tok->is(TT_ElseLBrace))
866 return true;
867 if (WrappedOpeningBrace) {
868 assert(Index > 0);
869 --Index; // The line above the wrapped l_brace.
870 Tok = nullptr;
871 }
872 return mightFitOnOneLine((*CurrentLines)[Index], Tok);
873 };
874 if (RemoveBraces()) {
875 Tok->MatchingParen = FormatTok;
876 FormatTok->MatchingParen = Tok;
877 }
878
879 size_t PPEndHash = computePPHash();
880
881 // Munch the closing brace.
882 nextToken(/*LevelDifference=*/-AddLevels);
883
884 // When this is a function block and there is an unnecessary semicolon
885 // afterwards then mark it as optional (so the RemoveSemi pass can get rid of
886 // it later).
887 if (Style.RemoveSemicolon && IsFunctionRBrace) {
888 while (FormatTok->is(tok::semi)) {
889 FormatTok->Optional = true;
890 nextToken();
891 }
892 }
893
894 HandleVerilogBlockLabel();
895
896 if (MacroBlock && FormatTok->is(tok::l_paren))
897 parseParens();
898
899 Line->Level = InitialLevel;
900
901 if (FormatTok->is(tok::kw_noexcept)) {
902 // A noexcept in a requires expression.
903 nextToken();
904 }
905
906 if (FormatTok->is(tok::arrow)) {
907 // Following the } or noexcept we can find a trailing return type arrow
908 // as part of an implicit conversion constraint.
909 nextToken();
910 parseStructuralElement();
911 }
912
913 if (MunchSemi && FormatTok->is(tok::semi))
914 nextToken();
915
916 if (PPStartHash == PPEndHash) {
917 Line->MatchingOpeningBlockLineIndex = OpeningLineIndex;
918 if (OpeningLineIndex != UnwrappedLine::kInvalidIndex) {
919 // Update the opening line to add the forward reference as well
920 (*CurrentLines)[OpeningLineIndex].MatchingClosingBlockLineIndex =
921 CurrentLines->size() - 1;
922 }
923 }
924
925 return IfLBrace;
926}
927
928static bool isGoogScope(const UnwrappedLine &Line) {
929 // FIXME: Closure-library specific stuff should not be hard-coded but be
930 // configurable.
931 if (Line.Tokens.size() < 4)
932 return false;
933 auto I = Line.Tokens.begin();
934 if (I->Tok->TokenText != "goog")
935 return false;
936 ++I;
937 if (I->Tok->isNot(tok::period))
938 return false;
939 ++I;
940 if (I->Tok->TokenText != "scope")
941 return false;
942 ++I;
943 return I->Tok->is(tok::l_paren);
944}
945
946static bool isIIFE(const UnwrappedLine &Line,
947 const AdditionalKeywords &Keywords) {
948 // Look for the start of an immediately invoked anonymous function.
949 // https://en.wikipedia.org/wiki/Immediately-invoked_function_expression
950 // This is commonly done in JavaScript to create a new, anonymous scope.
951 // Example: (function() { ... })()
952 if (Line.Tokens.size() < 3)
953 return false;
954 auto I = Line.Tokens.begin();
955 if (I->Tok->isNot(tok::l_paren))
956 return false;
957 ++I;
958 if (I->Tok->isNot(Keywords.kw_function))
959 return false;
960 ++I;
961 return I->Tok->is(tok::l_paren);
962}
963
964static bool ShouldBreakBeforeBrace(const FormatStyle &Style,
965 const FormatToken &InitialToken,
966 bool IsEmptyBlock,
967 bool IsJavaRecord = false) {
968 if (IsJavaRecord)
969 return Style.BraceWrapping.AfterClass;
970
971 tok::TokenKind Kind = InitialToken.Tok.getKind();
972 if (InitialToken.is(TT_NamespaceMacro))
973 Kind = tok::kw_namespace;
974
975 const bool WrapRecordAllowed =
976 !IsEmptyBlock ||
977 Style.AllowShortRecordOnASingleLine < FormatStyle::SRS_Empty ||
978 Style.BraceWrapping.SplitEmptyRecord;
979
980 switch (Kind) {
981 case tok::kw_namespace:
982 return Style.BraceWrapping.AfterNamespace;
983 case tok::kw_class:
984 return Style.BraceWrapping.AfterClass && WrapRecordAllowed;
985 case tok::kw_union:
986 return Style.BraceWrapping.AfterUnion && WrapRecordAllowed;
987 case tok::kw_struct:
988 return Style.BraceWrapping.AfterStruct && WrapRecordAllowed;
989 case tok::kw_enum:
990 return Style.BraceWrapping.AfterEnum;
991 default:
992 return false;
993 }
994}
995
996void UnwrappedLineParser::parseChildBlock() {
997 assert(FormatTok->is(tok::l_brace));
998 FormatTok->setBlockKind(BK_Block);
999 const FormatToken *OpeningBrace = FormatTok;
1000 nextToken();
1001 {
1002 bool SkipIndent = (Style.isJavaScript() &&
1003 (isGoogScope(*Line) || isIIFE(*Line, Keywords)));
1004 ScopedLineState LineState(*this);
1005 ScopedDeclarationState DeclarationState(*Line, DeclarationScopeStack,
1006 /*MustBeDeclaration=*/false);
1007 Line->Level += SkipIndent ? 0 : 1;
1008 parseLevel(OpeningBrace);
1009 flushComments(isOnNewLine(*FormatTok));
1010 Line->Level -= SkipIndent ? 0 : 1;
1011 }
1012 nextToken();
1013}
1014
1015void UnwrappedLineParser::parsePPDirective() {
1016 assert(FormatTok->is(tok::hash) && "'#' expected");
1017 ScopedMacroState MacroState(*Line, Tokens, FormatTok);
1018
1019 nextToken();
1020
1021 if (!FormatTok->Tok.getIdentifierInfo()) {
1022 parsePPUnknown();
1023 return;
1024 }
1025
1026 switch (FormatTok->Tok.getIdentifierInfo()->getPPKeywordID()) {
1027 case tok::pp_define:
1028 parsePPDefine();
1029 return;
1030 case tok::pp_if:
1031 parsePPIf(/*IfDef=*/false);
1032 break;
1033 case tok::pp_ifdef:
1034 case tok::pp_ifndef:
1035 parsePPIf(/*IfDef=*/true);
1036 break;
1037 case tok::pp_else:
1038 case tok::pp_elifdef:
1039 case tok::pp_elifndef:
1040 case tok::pp_elif:
1041 parsePPElse();
1042 break;
1043 case tok::pp_endif:
1044 parsePPEndIf();
1045 break;
1046 case tok::pp_pragma:
1047 parsePPPragma();
1048 break;
1049 case tok::pp_error:
1050 case tok::pp_warning:
1051 nextToken();
1052 if (!eof() && Style.isCpp())
1053 FormatTok->setFinalizedType(TT_AfterPPDirective);
1054 [[fallthrough]];
1055 default:
1056 parsePPUnknown();
1057 break;
1058 }
1059}
1060
1061void UnwrappedLineParser::conditionalCompilationCondition(bool Unreachable) {
1062 size_t Line = CurrentLines->size();
1063 if (CurrentLines == &PreprocessorDirectives)
1064 Line += Lines.size();
1065
1066 if (Unreachable ||
1067 (!PP.Stack.empty() && PP.Stack.back().Kind == PP_Unreachable)) {
1068 PP.Stack.push_back({PP_Unreachable, Line});
1069 } else {
1070 PP.Stack.push_back({PP_Conditional, Line});
1071 }
1072}
1073
1074void UnwrappedLineParser::conditionalCompilationStart(bool Unreachable) {
1075 ++PP.BranchLevel;
1076 assert(PP.BranchLevel >= 0 &&
1077 PP.BranchLevel <= (int)PP.LevelBranchIndex.size());
1078 if (PP.BranchLevel == (int)PP.LevelBranchIndex.size()) {
1079 PP.LevelBranchIndex.push_back(0);
1080 PP.LevelBranchCount.push_back(0);
1081 }
1082 PP.ChainBranchIndex.push(Unreachable ? -1 : 0);
1083 bool Skip = PP.LevelBranchIndex[PP.BranchLevel] > 0;
1084 conditionalCompilationCondition(Unreachable || Skip);
1085}
1086
1087void UnwrappedLineParser::conditionalCompilationAlternative() {
1088 if (!PP.Stack.empty())
1089 PP.Stack.pop_back();
1090 assert(PP.BranchLevel < (int)PP.LevelBranchIndex.size());
1091 if (!PP.ChainBranchIndex.empty())
1092 ++PP.ChainBranchIndex.top();
1093 conditionalCompilationCondition(
1094 PP.BranchLevel >= 0 && !PP.ChainBranchIndex.empty() &&
1095 PP.LevelBranchIndex[PP.BranchLevel] != PP.ChainBranchIndex.top());
1096}
1097
1098void UnwrappedLineParser::conditionalCompilationEnd() {
1099 assert(PP.BranchLevel < (int)PP.LevelBranchIndex.size());
1100 if (PP.BranchLevel >= 0 && !PP.ChainBranchIndex.empty()) {
1101 if (PP.ChainBranchIndex.top() + 1 > PP.LevelBranchCount[PP.BranchLevel])
1102 PP.LevelBranchCount[PP.BranchLevel] = PP.ChainBranchIndex.top() + 1;
1103 }
1104 // Guard against #endif's without #if.
1105 if (PP.BranchLevel > -1)
1106 --PP.BranchLevel;
1107 if (!PP.ChainBranchIndex.empty())
1108 PP.ChainBranchIndex.pop();
1109 if (!PP.Stack.empty())
1110 PP.Stack.pop_back();
1111}
1112
1113void UnwrappedLineParser::parsePPIf(bool IfDef) {
1114 bool IfNDef = FormatTok->is(tok::pp_ifndef);
1115 nextToken();
1116 bool Unreachable = false;
1117 if (!IfDef && (FormatTok->is(tok::kw_false) || FormatTok->TokenText == "0"))
1118 Unreachable = true;
1119 if (IfDef && !IfNDef && FormatTok->TokenText == "SWIG")
1120 Unreachable = true;
1121 conditionalCompilationStart(Unreachable);
1122 FormatToken *IfCondition = FormatTok;
1123 // If there's a #ifndef on the first line, and the only lines before it are
1124 // comments, it could be an include guard.
1125 bool MaybeIncludeGuard = IfNDef;
1126 if (PP.IncludeGuard == IG_Inited && MaybeIncludeGuard) {
1127 for (auto &Line : Lines) {
1128 if (Line.Tokens.front().Tok->isNot(tok::comment)) {
1129 MaybeIncludeGuard = false;
1130 PP.IncludeGuard = IG_Rejected;
1131 break;
1132 }
1133 }
1134 }
1135 --PP.BranchLevel;
1136 parsePPUnknown();
1137 ++PP.BranchLevel;
1138 if (PP.IncludeGuard == IG_Inited && MaybeIncludeGuard) {
1139 PP.IncludeGuard = IG_IfNdefed;
1140 PP.IncludeGuardToken = IfCondition;
1141 }
1142}
1143
1144void UnwrappedLineParser::parsePPElse() {
1145 // If a potential include guard has an #else, it's not an include guard.
1146 if (PP.IncludeGuard == IG_Defined && PP.BranchLevel == 0)
1147 PP.IncludeGuard = IG_Rejected;
1148 // Don't crash when there is an #else without an #if.
1149 assert(PP.BranchLevel >= -1);
1150 if (PP.BranchLevel == -1)
1151 conditionalCompilationStart(/*Unreachable=*/true);
1152 conditionalCompilationAlternative();
1153 --PP.BranchLevel;
1154 parsePPUnknown();
1155 ++PP.BranchLevel;
1156}
1157
1158void UnwrappedLineParser::parsePPEndIf() {
1159 conditionalCompilationEnd();
1160 parsePPUnknown();
1161}
1162
1163void UnwrappedLineParser::parsePPDefine() {
1164 nextToken();
1165
1166 if (!FormatTok->Tok.getIdentifierInfo()) {
1167 PP.IncludeGuard = IG_Rejected;
1168 PP.IncludeGuardToken = nullptr;
1169 parsePPUnknown();
1170 return;
1171 }
1172
1173 bool MaybeIncludeGuard = false;
1174 if (PP.IncludeGuard == IG_IfNdefed &&
1175 PP.IncludeGuardToken->TokenText == FormatTok->TokenText) {
1176 PP.IncludeGuard = IG_Defined;
1177 PP.IncludeGuardToken = nullptr;
1178 for (auto &Line : Lines) {
1179 if (Line.Tokens.front().Tok->isNoneOf(tok::comment, tok::hash)) {
1180 PP.IncludeGuard = IG_Rejected;
1181 break;
1182 }
1183 }
1184 MaybeIncludeGuard = PP.IncludeGuard == IG_Defined;
1185 }
1186
1187 // In the context of a define, even keywords should be treated as normal
1188 // identifiers. Setting the kind to identifier is not enough, because we need
1189 // to treat additional keywords like __except as well, which are already
1190 // identifiers. Setting the identifier info to null interferes with include
1191 // guard processing above, and changes preprocessing nesting.
1192 FormatTok->Tok.setKind(tok::identifier);
1193 FormatTok->Tok.setIdentifierInfo(Keywords.kw_internal_ident_after_define);
1194 nextToken();
1195
1196 // IncludeGuard can't have a non-empty macro definition.
1197 if (MaybeIncludeGuard && !eof())
1198 PP.IncludeGuard = IG_Rejected;
1199
1200 if (FormatTok->is(tok::l_paren) && !FormatTok->hasWhitespaceBefore())
1201 parseParens();
1202 if (Style.IndentPPDirectives != FormatStyle::PPDIS_None)
1203 Line->Level += PP.BranchLevel + 1;
1204 addUnwrappedLine();
1205 ++Line->Level;
1206
1207 Line->PPLevel = PP.BranchLevel + (PP.IncludeGuard == IG_Defined ? 0 : 1);
1208 assert((int)Line->PPLevel >= 0);
1209
1210 if (eof())
1211 return;
1212
1213 Line->InMacroBody = true;
1214
1215 if (!Style.SkipMacroDefinitionBody) {
1216 // Errors during a preprocessor directive can only affect the layout of the
1217 // preprocessor directive, and thus we ignore them. An alternative approach
1218 // would be to use the same approach we use on the file level (no
1219 // re-indentation if there was a structural error) within the macro
1220 // definition.
1221 parseFile();
1222 return;
1223 }
1224
1225 for (auto *Comment : CommentsBeforeNextToken)
1226 Comment->Finalized = true;
1227
1228 do {
1229 FormatTok->Finalized = true;
1230 FormatTok = Tokens->getNextToken();
1231 } while (!eof());
1232
1233 addUnwrappedLine();
1234}
1235
1236void UnwrappedLineParser::parsePPPragma() {
1237 Line->InPragmaDirective = true;
1238 parsePPUnknown();
1239}
1240
1241void UnwrappedLineParser::parsePPUnknown() {
1242 while (!eof())
1243 nextToken();
1244 if (Style.IndentPPDirectives != FormatStyle::PPDIS_None)
1245 Line->Level += PP.BranchLevel + 1;
1246 addUnwrappedLine();
1247}
1248
1249// Here we exclude certain tokens that are not usually the first token in an
1250// unwrapped line. This is used in attempt to distinguish macro calls without
1251// trailing semicolons from other constructs split to several lines.
1253 // Semicolon can be a null-statement, l_square can be a start of a macro or
1254 // a C++11 attribute, but this doesn't seem to be common.
1255 return Tok.isNoneOf(tok::semi, tok::l_brace,
1256 // Tokens that can only be used as binary operators and a
1257 // part of overloaded operator names.
1258 tok::period, tok::periodstar, tok::arrow, tok::arrowstar,
1259 tok::less, tok::greater, tok::slash, tok::percent,
1260 tok::lessless, tok::greatergreater, tok::equal,
1261 tok::plusequal, tok::minusequal, tok::starequal,
1262 tok::slashequal, tok::percentequal, tok::ampequal,
1263 tok::pipeequal, tok::caretequal, tok::greatergreaterequal,
1264 tok::lesslessequal,
1265 // Colon is used in labels, base class lists, initializer
1266 // lists, range-based for loops, ternary operator, but
1267 // should never be the first token in an unwrapped line.
1268 tok::colon,
1269 // 'noexcept' is a trailing annotation.
1270 tok::kw_noexcept);
1271}
1272
1273static bool mustBeJSIdent(const AdditionalKeywords &Keywords,
1274 const FormatToken *FormatTok) {
1275 // FIXME: This returns true for C/C++ keywords like 'struct'.
1276 return FormatTok->is(tok::identifier) &&
1277 (!FormatTok->Tok.getIdentifierInfo() ||
1278 FormatTok->isNoneOf(
1279 Keywords.kw_in, Keywords.kw_of, Keywords.kw_as, Keywords.kw_async,
1280 Keywords.kw_await, Keywords.kw_yield, Keywords.kw_finally,
1281 Keywords.kw_function, Keywords.kw_import, Keywords.kw_is,
1282 Keywords.kw_let, Keywords.kw_var, tok::kw_const,
1283 Keywords.kw_abstract, Keywords.kw_extends, Keywords.kw_implements,
1284 Keywords.kw_instanceof, Keywords.kw_interface,
1285 Keywords.kw_override, Keywords.kw_throws, Keywords.kw_from));
1286}
1287
1288static bool mustBeJSIdentOrValue(const AdditionalKeywords &Keywords,
1289 const FormatToken *FormatTok) {
1290 return FormatTok->Tok.isLiteral() ||
1291 FormatTok->isOneOf(tok::kw_true, tok::kw_false) ||
1292 mustBeJSIdent(Keywords, FormatTok);
1293}
1294
1295// isJSDeclOrStmt returns true if |FormatTok| starts a declaration or statement
1296// when encountered after a value (see mustBeJSIdentOrValue).
1297static bool isJSDeclOrStmt(const AdditionalKeywords &Keywords,
1298 const FormatToken *FormatTok) {
1299 return FormatTok->isOneOf(
1300 tok::kw_return, Keywords.kw_yield,
1301 // conditionals
1302 tok::kw_if, tok::kw_else,
1303 // loops
1304 tok::kw_for, tok::kw_while, tok::kw_do, tok::kw_continue, tok::kw_break,
1305 // switch/case
1306 tok::kw_switch, tok::kw_case,
1307 // exceptions
1308 tok::kw_throw, tok::kw_try, tok::kw_catch, Keywords.kw_finally,
1309 // declaration
1310 tok::kw_const, tok::kw_class, Keywords.kw_var, Keywords.kw_let,
1311 Keywords.kw_async, Keywords.kw_function,
1312 // import/export
1313 Keywords.kw_import, tok::kw_export);
1314}
1315
1316// Checks whether a token is a type in K&R C (aka C78).
1317static bool isC78Type(const FormatToken &Tok) {
1318 return Tok.isOneOf(tok::kw_char, tok::kw_short, tok::kw_int, tok::kw_long,
1319 tok::kw_unsigned, tok::kw_float, tok::kw_double,
1320 tok::identifier);
1321}
1322
1323// This function checks whether a token starts the first parameter declaration
1324// in a K&R C (aka C78) function definition, e.g.:
1325// int f(a, b)
1326// short a, b;
1327// {
1328// return a + b;
1329// }
1331 const FormatToken *FuncName) {
1332 assert(Tok);
1333 assert(Next);
1334 assert(FuncName);
1335
1336 if (FuncName->isNot(tok::identifier))
1337 return false;
1338
1339 const FormatToken *Prev = FuncName->Previous;
1340 if (!Prev || (Prev->isNot(tok::star) && !isC78Type(*Prev)))
1341 return false;
1342
1343 if (!isC78Type(*Tok) &&
1344 Tok->isNoneOf(tok::kw_register, tok::kw_struct, tok::kw_union)) {
1345 return false;
1346 }
1347
1348 if (Next->isNot(tok::star) && !Next->Tok.getIdentifierInfo())
1349 return false;
1350
1351 Tok = Tok->Previous;
1352 if (!Tok || Tok->isNot(tok::r_paren))
1353 return false;
1354
1355 Tok = Tok->Previous;
1356 if (!Tok || Tok->isNot(tok::identifier))
1357 return false;
1358
1359 return Tok->Previous && Tok->Previous->isOneOf(tok::l_paren, tok::comma);
1360}
1361
1362bool UnwrappedLineParser::parseModuleDecl() {
1363 assert(IsCpp);
1364 assert(FormatTok->is(Keywords.kw_module));
1365
1366 if (Style.Language == FormatStyle::LK_C ||
1367 Style.Standard < FormatStyle::LS_Cpp20) {
1368 return false;
1369 }
1370
1371 nextToken();
1372 if (FormatTok->isNot(tok::identifier))
1373 return false;
1374
1375 for (nextToken(); FormatTok->isNoneOf(tok::semi, tok::eof); nextToken())
1376 if (FormatTok->is(tok::colon))
1377 FormatTok->setFinalizedType(TT_ModulePartitionColon);
1378
1379 nextToken();
1380 Line->IsModuleOrImportDecl = true;
1381 addUnwrappedLine();
1382 return true;
1383}
1384
1385bool UnwrappedLineParser::parseImportDecl() {
1386 assert(IsCpp);
1387 assert(FormatTok->is(Keywords.kw_import) && "'import' expected");
1388
1389 if (Style.Language == FormatStyle::LK_C ||
1390 Style.Standard < FormatStyle::LS_Cpp20) {
1391 return false;
1392 }
1393
1394 nextToken();
1395 if (FormatTok->is(tok::colon)) {
1396 FormatTok->setFinalizedType(TT_ModulePartitionColon);
1397 nextToken();
1398 }
1399 if (FormatTok->isNoneOf(tok::identifier, tok::less, tok::string_literal))
1400 return false;
1401
1402 for (; FormatTok->isNoneOf(tok::semi, tok::eof); nextToken()) {
1403 // Handle import <foo/bar.h> as we would an include statement.
1404 if (FormatTok->is(tok::less)) {
1405 for (nextToken(); FormatTok->isNoneOf(tok::greater, tok::semi, tok::eof);
1406 nextToken()) {
1407 // Mark tokens as implicit string literals, so that import <A/Foo> will
1408 // neither be broken nor have a space added.
1409 FormatTok->setFinalizedType(TT_ImplicitStringLiteral);
1410 }
1411 }
1412 }
1413
1414 nextToken();
1415 Line->IsModuleOrImportDecl = true;
1416 addUnwrappedLine();
1417 return true;
1418}
1419
1420// readTokenWithJavaScriptASI reads the next token and terminates the current
1421// line if JavaScript Automatic Semicolon Insertion must
1422// happen between the current token and the next token.
1423//
1424// This method is conservative - it cannot cover all edge cases of JavaScript,
1425// but only aims to correctly handle certain well known cases. It *must not*
1426// return true in speculative cases.
1427void UnwrappedLineParser::readTokenWithJavaScriptASI() {
1428 FormatToken *Previous = FormatTok;
1429 readToken();
1430 FormatToken *Next = FormatTok;
1431
1432 bool IsOnSameLine =
1433 CommentsBeforeNextToken.empty()
1434 ? Next->NewlinesBefore == 0
1435 : CommentsBeforeNextToken.front()->NewlinesBefore == 0;
1436 if (IsOnSameLine)
1437 return;
1438
1439 bool PreviousMustBeValue = mustBeJSIdentOrValue(Keywords, Previous);
1440 bool PreviousStartsTemplateExpr =
1441 Previous->is(TT_TemplateString) && Previous->TokenText.ends_with("${");
1442 if (PreviousMustBeValue || Previous->is(tok::r_paren)) {
1443 // If the line contains an '@' sign, the previous token might be an
1444 // annotation, which can precede another identifier/value.
1445 bool HasAt = llvm::any_of(Line->Tokens, [](UnwrappedLineNode &LineNode) {
1446 return LineNode.Tok->is(tok::at);
1447 });
1448 if (HasAt)
1449 return;
1450 }
1451 if (Next->is(tok::exclaim) && PreviousMustBeValue)
1452 return addUnwrappedLine();
1453 bool NextMustBeValue = mustBeJSIdentOrValue(Keywords, Next);
1454 bool NextEndsTemplateExpr =
1455 Next->is(TT_TemplateString) && Next->TokenText.starts_with("}");
1456 if (NextMustBeValue && !NextEndsTemplateExpr && !PreviousStartsTemplateExpr &&
1457 (PreviousMustBeValue ||
1458 Previous->isOneOf(tok::r_square, tok::r_paren, tok::plusplus,
1459 tok::minusminus))) {
1460 return addUnwrappedLine();
1461 }
1462 if ((PreviousMustBeValue || Previous->is(tok::r_paren)) &&
1463 isJSDeclOrStmt(Keywords, Next)) {
1464 return addUnwrappedLine();
1465 }
1466}
1467
1468void UnwrappedLineParser::parseStructuralElement(
1469 const FormatToken *OpeningBrace, IfStmtKind *IfKind,
1470 FormatToken **IfLeftBrace, bool *HasDoWhile, bool *HasLabel) {
1471 if (Style.isTableGen() && FormatTok->is(tok::pp_include)) {
1472 nextToken();
1473 if (FormatTok->is(tok::string_literal))
1474 nextToken();
1475 addUnwrappedLine();
1476 return;
1477 }
1478
1479 if (IsCpp) {
1480 while (FormatTok->is(tok::l_square) && handleCppAttributes()) {
1481 }
1482 } else if (Style.isVerilog()) {
1483 // Skip attributes.
1484 while (FormatTok->is(tok::l_paren) &&
1485 Tokens->peekNextToken()->is(tok::star)) {
1486 parseParens();
1487 }
1488 skipVerilogQualifiers();
1489 // Skip things that can exist before keywords like 'if' and 'case'.
1490 if (FormatTok->isOneOf(Keywords.kw_priority, Keywords.kw_unique,
1491 Keywords.kw_unique0)) {
1492 nextToken();
1493 }
1494
1495 if (Keywords.isVerilogStructuredProcedure(*FormatTok)) {
1496 parseForOrWhileLoop(/*HasParens=*/false);
1497 return;
1498 }
1499 if (FormatTok->isOneOf(Keywords.kw_foreach, Keywords.kw_repeat)) {
1500 parseForOrWhileLoop();
1501 return;
1502 }
1503 if (FormatTok->isOneOf(tok::kw_restrict, Keywords.kw_assert,
1504 Keywords.kw_assume, Keywords.kw_cover)) {
1505 parseIfThenElse(IfKind, /*KeepBraces=*/false, /*IsVerilogAssert=*/true);
1506 return;
1507 }
1508 }
1509
1510 // Tokens that only make sense at the beginning of a line.
1511 if (FormatTok->isAccessSpecifierKeyword()) {
1512 if (Style.isJava() || Style.isJavaScript() || Style.isCSharp())
1513 nextToken();
1514 else
1515 parseAccessSpecifier();
1516 return;
1517 }
1518 switch (FormatTok->Tok.getKind()) {
1519 case tok::kw_asm: {
1520 // Track whether to skip formatting inline asm by finalizing the tokens
1521 // in the block. Formatting is skipped inside of braces by default.
1522 // A style option could be added to also skip formatting inside parens.
1523 bool DoNotFormat = false;
1524 tok::TokenKind OpenType;
1525 tok::TokenKind CloseType;
1526 nextToken();
1527 while (FormatTok &&
1528 FormatTok->isOneOf(tok::kw_volatile, tok::kw_inline, tok::kw_goto)) {
1529 nextToken();
1530 }
1531 if (!FormatTok)
1532 break;
1533 if (FormatTok->is(tok::l_brace)) {
1534 FormatTok->setFinalizedType(TT_InlineASMBrace);
1535 OpenType = tok::l_brace;
1536 CloseType = tok::r_brace;
1537 DoNotFormat = true;
1538 } else if (FormatTok->is(tok::l_paren)) {
1539 OpenType = tok::l_paren;
1540 CloseType = tok::r_paren;
1541 FormatTok->setFinalizedType(TT_InlineASMParen);
1542 } else {
1543 break;
1544 }
1545 if (DoNotFormat) {
1546 FormatToken *OpenTok = FormatTok;
1547 int NestLevel = 0;
1548 nextToken();
1549 while (FormatTok && !eof()) {
1550 if (FormatTok->is(OpenType)) {
1551 ++NestLevel;
1552 } else if (FormatTok->is(CloseType)) {
1553 --NestLevel;
1554 if (NestLevel < 1) {
1555 FormatTok->setFinalizedType(OpenTok->getType());
1556 nextToken();
1557 addUnwrappedLine();
1558 break;
1559 }
1560 }
1561 FormatTok->Finalized = true;
1562 nextToken();
1563 }
1564 }
1565 break;
1566 }
1567 case tok::kw_namespace:
1568 parseNamespace();
1569 return;
1570 case tok::kw_if: {
1571 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1572 // field/method declaration.
1573 break;
1574 }
1575 FormatToken *Tok = parseIfThenElse(IfKind);
1576 if (IfLeftBrace)
1577 *IfLeftBrace = Tok;
1578 return;
1579 }
1580 case tok::kw_for:
1581 case tok::kw_while:
1582 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1583 // field/method declaration.
1584 break;
1585 }
1586 parseForOrWhileLoop();
1587 return;
1588 case tok::kw_do:
1589 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1590 // field/method declaration.
1591 break;
1592 }
1593 parseDoWhile();
1594 if (HasDoWhile)
1595 *HasDoWhile = true;
1596 return;
1597 case tok::kw_switch:
1598 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1599 // 'switch: string' field declaration.
1600 break;
1601 }
1602 parseSwitch(/*IsExpr=*/false);
1603 return;
1604 case tok::kw_default: {
1605 // In Verilog default along with other labels are handled in the next loop.
1606 if (Style.isVerilog())
1607 break;
1608 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1609 // 'default: string' field declaration.
1610 break;
1611 }
1612 auto *Default = FormatTok;
1613 nextToken();
1614 if (FormatTok->is(tok::colon)) {
1615 FormatTok->setFinalizedType(TT_CaseLabelColon);
1616 parseLabel();
1617 return;
1618 }
1619 if (FormatTok->is(tok::arrow)) {
1620 FormatTok->setFinalizedType(TT_CaseLabelArrow);
1621 Default->setFinalizedType(TT_SwitchExpressionLabel);
1622 parseLabel();
1623 return;
1624 }
1625 // e.g. "default void f() {}" in a Java interface.
1626 break;
1627 }
1628 case tok::kw_case:
1629 // Proto: there are no switch/case statements.
1630 if (Style.Language == FormatStyle::LK_Proto) {
1631 nextToken();
1632 return;
1633 }
1634 if (Style.isVerilog()) {
1635 parseBlock();
1636 addUnwrappedLine();
1637 return;
1638 }
1639 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1640 // 'case: string' field declaration.
1641 nextToken();
1642 break;
1643 }
1644 parseCaseLabel();
1645 return;
1646 case tok::kw_goto:
1647 nextToken();
1648 if (FormatTok->is(tok::kw_case))
1649 nextToken();
1650 break;
1651 case tok::kw_try:
1652 case tok::kw___try:
1653 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1654 // field/method declaration.
1655 break;
1656 }
1657 parseTryCatch();
1658 return;
1659 case tok::kw_extern:
1660 if (Style.isVerilog()) {
1661 // In Verilog an extern module declaration looks like a start of module.
1662 // But there is no body and endmodule. So we handle it separately.
1663 parseVerilogExtern();
1664 return;
1665 }
1666 nextToken();
1667 if (FormatTok->is(tok::string_literal)) {
1668 nextToken();
1669 if (FormatTok->is(tok::l_brace)) {
1670 if (Style.BraceWrapping.AfterExternBlock)
1671 addUnwrappedLine();
1672 // Either we indent or for backwards compatibility we follow the
1673 // AfterExternBlock style.
1674 unsigned AddLevels =
1675 (Style.IndentExternBlock == FormatStyle::IEBS_Indent) ||
1676 (Style.BraceWrapping.AfterExternBlock &&
1677 Style.IndentExternBlock ==
1679 ? 1u
1680 : 0u;
1681 parseBlock(/*MustBeDeclaration=*/true, AddLevels);
1682 addUnwrappedLine();
1683 return;
1684 }
1685 }
1686 break;
1687 case tok::kw_export:
1688 if (IsCpp) {
1689 nextToken();
1690 if (FormatTok->is(tok::kw_namespace)) {
1691 parseNamespace();
1692 return;
1693 }
1694 if (FormatTok->is(tok::l_brace)) {
1695 parseCppExportBlock();
1696 return;
1697 }
1698 if (FormatTok->is(Keywords.kw_module) && parseModuleDecl())
1699 return;
1700 if (FormatTok->is(Keywords.kw_import) && parseImportDecl())
1701 return;
1702 break;
1703 }
1704 if (Style.isJavaScript()) {
1705 parseJavaScriptEs6ImportExport();
1706 return;
1707 }
1708 if (Style.isVerilog()) {
1709 parseVerilogExtern();
1710 return;
1711 }
1712 break;
1713 case tok::kw_inline:
1714 nextToken();
1715 if (FormatTok->is(tok::kw_namespace)) {
1716 parseNamespace();
1717 return;
1718 }
1719 break;
1720 case tok::identifier:
1721 if (FormatTok->is(TT_ForEachMacro)) {
1722 parseForOrWhileLoop();
1723 return;
1724 }
1725 if (FormatTok->is(TT_MacroBlockBegin)) {
1726 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
1727 /*MunchSemi=*/false);
1728 return;
1729 }
1730 if (FormatTok->is(Keywords.kw_import)) {
1731 if (IsCpp && parseImportDecl())
1732 return;
1733 if (Style.isJavaScript()) {
1734 parseJavaScriptEs6ImportExport();
1735 return;
1736 }
1737 if (Style.Language == FormatStyle::LK_Proto) {
1738 nextToken();
1739 if (FormatTok->is(tok::kw_public))
1740 nextToken();
1741 if (FormatTok->isNot(tok::string_literal))
1742 return;
1743 nextToken();
1744 if (FormatTok->is(tok::semi))
1745 nextToken();
1746 addUnwrappedLine();
1747 return;
1748 }
1749 if (Style.isVerilog()) {
1750 parseVerilogExtern();
1751 return;
1752 }
1753 }
1754 if (IsCpp) {
1755 if (FormatTok->is(Keywords.kw_module) && parseModuleDecl())
1756 return;
1757 if (FormatTok->isOneOf(Keywords.kw_signals, Keywords.kw_qsignals,
1758 Keywords.kw_slots, Keywords.kw_qslots)) {
1759 nextToken();
1760 if (FormatTok->is(tok::colon)) {
1761 nextToken();
1762 addUnwrappedLine();
1763 return;
1764 }
1765 }
1766 if (FormatTok->is(TT_StatementMacro)) {
1767 parseStatementMacro();
1768 return;
1769 }
1770 if (FormatTok->is(TT_NamespaceMacro)) {
1771 parseNamespace();
1772 return;
1773 }
1774 }
1775 // In Verilog labels can be any expression, so we don't do them here.
1776 // JS doesn't have macros, and within classes colons indicate fields, not
1777 // labels.
1778 // TableGen doesn't have labels.
1779 if (!Style.isJavaScript() && !Style.isVerilog() && !Style.isTableGen() &&
1780 Tokens->peekNextToken()->is(tok::colon) && !Line->MustBeDeclaration) {
1781 nextToken();
1782 if (!Line->InMacroBody || CurrentLines->size() > 1)
1783 Line->Tokens.begin()->Tok->MustBreakBefore = true;
1784 FormatTok->setFinalizedType(TT_GotoLabelColon);
1785 parseLabel(/*IsGotoLabel=*/true);
1786 if (HasLabel)
1787 *HasLabel = true;
1788 return;
1789 }
1790 if (Style.isJava() && FormatTok->is(Keywords.kw_record)) {
1791 parseRecord(/*ParseAsExpr=*/false, /*IsJavaRecord=*/true);
1792 addUnwrappedLine();
1793 return;
1794 }
1795 // In all other cases, parse the declaration.
1796 break;
1797 default:
1798 break;
1799 }
1800
1801 bool SeenEqual = false;
1802 for (const bool InRequiresExpression =
1803 OpeningBrace && OpeningBrace->isOneOf(TT_RequiresExpressionLBrace,
1804 TT_CompoundRequirementLBrace);
1805 !eof();) {
1806 const FormatToken *Previous = FormatTok->Previous;
1807 switch (FormatTok->Tok.getKind()) {
1808 case tok::at:
1809 nextToken();
1810 if (FormatTok->is(tok::l_brace)) {
1811 nextToken();
1812 parseBracedList();
1813 break;
1814 }
1815 if (Style.isJava() && FormatTok->is(Keywords.kw_interface)) {
1816 nextToken();
1817 break;
1818 }
1819 switch (bool IsAutoRelease = false; FormatTok->Tok.getObjCKeywordID()) {
1820 case tok::objc_public:
1821 case tok::objc_protected:
1822 case tok::objc_package:
1823 case tok::objc_private:
1824 return parseAccessSpecifier();
1825 case tok::objc_interface:
1826 case tok::objc_implementation:
1827 return parseObjCInterfaceOrImplementation();
1828 case tok::objc_protocol:
1829 if (parseObjCProtocol())
1830 return;
1831 break;
1832 case tok::objc_end:
1833 return; // Handled by the caller.
1834 case tok::objc_optional:
1835 case tok::objc_required:
1836 nextToken();
1837 addUnwrappedLine();
1838 return;
1839 case tok::objc_autoreleasepool:
1840 IsAutoRelease = true;
1841 [[fallthrough]];
1842 case tok::objc_synchronized:
1843 nextToken();
1844 if (!IsAutoRelease && FormatTok->is(tok::l_paren)) {
1845 // Skip synchronization object
1846 parseParens();
1847 }
1848 if (FormatTok->is(tok::l_brace)) {
1849 if (Style.BraceWrapping.AfterControlStatement ==
1851 addUnwrappedLine();
1852 }
1853 parseBlock();
1854 }
1855 addUnwrappedLine();
1856 return;
1857 case tok::objc_try:
1858 // This branch isn't strictly necessary (the kw_try case below would
1859 // do this too after the tok::at is parsed above). But be explicit.
1860 parseTryCatch();
1861 return;
1862 default:
1863 break;
1864 }
1865 break;
1866 case tok::kw_requires: {
1867 if (IsCpp) {
1868 bool ParsedClause = parseRequires(SeenEqual);
1869 if (ParsedClause)
1870 return;
1871 } else {
1872 nextToken();
1873 }
1874 break;
1875 }
1876 case tok::kw_enum:
1877 // Ignore if this is part of "template <enum ..." or "... -> enum" or
1878 // "template <..., enum ...>".
1879 if (Previous && Previous->isOneOf(tok::less, tok::arrow, tok::comma)) {
1880 nextToken();
1881 break;
1882 }
1883
1884 // parseEnum falls through and does not yet add an unwrapped line as an
1885 // enum definition can start a structural element.
1886 if (!parseEnum())
1887 break;
1888 // This only applies to C++ and Verilog.
1889 if (!IsCpp && !Style.isVerilog()) {
1890 addUnwrappedLine();
1891 return;
1892 }
1893 break;
1894 case tok::kw_typedef:
1895 nextToken();
1896 if (FormatTok->isOneOf(Keywords.kw_NS_ENUM, Keywords.kw_NS_OPTIONS,
1897 Keywords.kw_CF_ENUM, Keywords.kw_CF_OPTIONS,
1898 Keywords.kw_CF_CLOSED_ENUM,
1899 Keywords.kw_NS_CLOSED_ENUM)) {
1900 parseEnum();
1901 }
1902 break;
1903 case tok::kw_class:
1904 if (Style.isVerilog()) {
1905 parseBlock();
1906 addUnwrappedLine();
1907 return;
1908 }
1909 if (Style.isTableGen()) {
1910 // Do nothing special. In this case the l_brace becomes FunctionLBrace.
1911 // This is same as def and so on.
1912 nextToken();
1913 break;
1914 }
1915 [[fallthrough]];
1916 case tok::kw_struct:
1917 case tok::kw_union:
1918 if (parseStructLike())
1919 return;
1920 break;
1921 case tok::kw_decltype:
1922 nextToken();
1923 if (FormatTok->is(tok::l_paren)) {
1924 parseParens();
1925 if (FormatTok->Previous &&
1926 FormatTok->Previous->endsSequence(tok::r_paren, tok::kw_auto,
1927 tok::l_paren)) {
1928 Line->SeenDecltypeAuto = true;
1929 }
1930 }
1931 break;
1932 case tok::period:
1933 nextToken();
1934 // In Java, classes have an implicit static member "class".
1935 if (Style.isJava() && FormatTok && FormatTok->is(tok::kw_class))
1936 nextToken();
1937 if (Style.isJavaScript() && FormatTok &&
1938 FormatTok->Tok.getIdentifierInfo()) {
1939 // JavaScript only has pseudo keywords, all keywords are allowed to
1940 // appear in "IdentifierName" positions. See http://es5.github.io/#x7.6
1941 nextToken();
1942 }
1943 break;
1944 case tok::semi:
1945 nextToken();
1946 addUnwrappedLine();
1947 return;
1948 case tok::r_brace:
1949 addUnwrappedLine();
1950 return;
1951 case tok::string_literal:
1952 if (Style.isVerilog() && FormatTok->is(TT_VerilogProtected)) {
1953 FormatTok->Finalized = true;
1954 nextToken();
1955 addUnwrappedLine();
1956 return;
1957 }
1958 nextToken();
1959 break;
1960 case tok::l_paren: {
1961 parseParens();
1962 // Break the unwrapped line if a K&R C function definition has a parameter
1963 // declaration.
1964 if (OpeningBrace || !IsCpp || !Previous || eof())
1965 break;
1966 if (isC78ParameterDecl(FormatTok,
1967 Tokens->peekNextToken(/*SkipComment=*/true),
1968 Previous)) {
1969 addUnwrappedLine();
1970 return;
1971 }
1972 break;
1973 }
1974 case tok::kw_operator:
1975 nextToken();
1976 if (FormatTok->isBinaryOperator())
1977 nextToken();
1978 break;
1979 case tok::caret: {
1980 const auto *Prev = FormatTok->getPreviousNonComment();
1981 nextToken();
1982 if (Prev && Prev->is(tok::identifier))
1983 break;
1984 // Block return type.
1985 if (FormatTok->Tok.isAnyIdentifier() || FormatTok->isTypeName(LangOpts)) {
1986 nextToken();
1987 // Return types: ObjC generics and protocol qualifiers are ok too.
1988 if (FormatTok->is(tok::less)) {
1989 nextToken();
1990 parseBracedList(/*IsAngleBracket=*/true);
1991 }
1992 // Return types: pointers are ok too.
1993 while (FormatTok->is(tok::star))
1994 nextToken();
1995 }
1996 // Block argument list.
1997 if (FormatTok->is(tok::l_paren))
1998 parseParens();
1999 // Block body.
2000 if (FormatTok->is(tok::l_brace))
2001 parseChildBlock();
2002 break;
2003 }
2004 case tok::l_brace:
2005 if (InRequiresExpression)
2006 FormatTok->setFinalizedType(TT_BracedListLBrace);
2007 if (!tryToParsePropertyAccessor() && !tryToParseBracedList()) {
2008 IsDecltypeAutoFunction = Line->SeenDecltypeAuto;
2009 // A block outside of parentheses must be the last part of a
2010 // structural element.
2011 // FIXME: Figure out cases where this is not true, and add projections
2012 // for them (the one we know is missing are lambdas).
2013 if (Style.isJava() &&
2014 Line->Tokens.front().Tok->is(Keywords.kw_synchronized)) {
2015 // If necessary, we could set the type to something different than
2016 // TT_FunctionLBrace.
2017 if (Style.BraceWrapping.AfterControlStatement ==
2019 addUnwrappedLine();
2020 }
2021 } else if (Style.BraceWrapping.AfterFunction) {
2022 addUnwrappedLine();
2023 }
2024 if (!Previous || Previous->isNot(TT_TypeDeclarationParen))
2025 FormatTok->setFinalizedType(TT_FunctionLBrace);
2026 parseBlock();
2027 IsDecltypeAutoFunction = false;
2028 addUnwrappedLine();
2029 return;
2030 }
2031 // Otherwise this was a braced init list, and the structural
2032 // element continues.
2033 break;
2034 case tok::kw_try:
2035 if (Style.isJavaScript() && Line->MustBeDeclaration) {
2036 // field/method declaration.
2037 nextToken();
2038 break;
2039 }
2040 // We arrive here when parsing function-try blocks.
2041 if (Style.BraceWrapping.AfterFunction)
2042 addUnwrappedLine();
2043 parseTryCatch();
2044 return;
2045 case tok::identifier: {
2046 if (Style.isCSharp() && FormatTok->is(Keywords.kw_where) &&
2047 Line->MustBeDeclaration) {
2048 addUnwrappedLine();
2049 parseCSharpGenericTypeConstraint();
2050 break;
2051 }
2052 if (FormatTok->is(TT_MacroBlockEnd)) {
2053 addUnwrappedLine();
2054 return;
2055 }
2056
2057 // Function declarations (as opposed to function expressions) are parsed
2058 // on their own unwrapped line by continuing this loop. Function
2059 // expressions (functions that are not on their own line) must not create
2060 // a new unwrapped line, so they are special cased below.
2061 size_t TokenCount = Line->Tokens.size();
2062 if (Style.isJavaScript() && FormatTok->is(Keywords.kw_function) &&
2063 (TokenCount > 1 ||
2064 (TokenCount == 1 &&
2065 Line->Tokens.front().Tok->isNot(Keywords.kw_async)))) {
2066 tryToParseJSFunction();
2067 break;
2068 }
2069 if ((Style.isJavaScript() || Style.isJava()) &&
2070 FormatTok->is(Keywords.kw_interface)) {
2071 if (Style.isJavaScript()) {
2072 // In JavaScript/TypeScript, "interface" can be used as a standalone
2073 // identifier, e.g. in `var interface = 1;`. If "interface" is
2074 // followed by another identifier, it is very like to be an actual
2075 // interface declaration.
2076 unsigned StoredPosition = Tokens->getPosition();
2077 FormatToken *Next = Tokens->getNextToken();
2078 FormatTok = Tokens->setPosition(StoredPosition);
2079 if (!mustBeJSIdent(Keywords, Next)) {
2080 nextToken();
2081 break;
2082 }
2083 }
2084 parseRecord();
2085 addUnwrappedLine();
2086 return;
2087 }
2088
2089 if (Style.isVerilog()) {
2090 if (FormatTok->is(Keywords.kw_table)) {
2091 parseVerilogTable();
2092 return;
2093 }
2094 if (Keywords.isVerilogBegin(*FormatTok) ||
2095 Keywords.isVerilogHierarchy(*FormatTok)) {
2096 parseBlock();
2097 addUnwrappedLine();
2098 return;
2099 }
2100 }
2101
2102 if (!IsCpp && FormatTok->is(Keywords.kw_interface)) {
2103 if (parseStructLike())
2104 return;
2105 break;
2106 }
2107
2108 if (IsCpp && FormatTok->is(TT_StatementMacro)) {
2109 parseStatementMacro();
2110 return;
2111 }
2112
2113 // See if the following token should start a new unwrapped line.
2114 StringRef Text = FormatTok->TokenText;
2115
2116 FormatToken *PreviousToken = FormatTok;
2117 nextToken();
2118
2119 // JS doesn't have macros, and within classes colons indicate fields, not
2120 // labels.
2121 if (Style.isJavaScript())
2122 break;
2123
2124 auto OneTokenSoFar = [&]() {
2125 auto I = Line->Tokens.begin(), E = Line->Tokens.end();
2126 while (I != E && I->Tok->is(tok::comment))
2127 ++I;
2128 if (Style.isVerilog())
2129 while (I != E && I->Tok->is(tok::hash))
2130 ++I;
2131 return I != E && (++I == E);
2132 };
2133 if (OneTokenSoFar()) {
2134 // Recognize function-like macro usages without trailing semicolon as
2135 // well as free-standing macros like Q_OBJECT.
2136 bool FunctionLike = FormatTok->is(tok::l_paren);
2137 if (FunctionLike)
2138 parseParens();
2139
2140 bool FollowedByNewline =
2141 CommentsBeforeNextToken.empty()
2142 ? FormatTok->NewlinesBefore > 0
2143 : CommentsBeforeNextToken.front()->NewlinesBefore > 0;
2144
2145 if (FollowedByNewline &&
2146 (Text.size() >= 5 ||
2147 (FunctionLike && FormatTok->isNot(tok::l_paren))) &&
2148 tokenCanStartNewLine(*FormatTok) && Text == Text.upper()) {
2149 if (PreviousToken->isNot(TT_UntouchableMacroFunc))
2150 PreviousToken->setFinalizedType(TT_FunctionLikeOrFreestandingMacro);
2151 addUnwrappedLine();
2152 return;
2153 }
2154 }
2155 break;
2156 }
2157 case tok::equal:
2158 if ((Style.isJavaScript() || Style.isCSharp()) &&
2159 FormatTok->is(TT_FatArrow)) {
2160 tryToParseChildBlock();
2161 break;
2162 }
2163
2164 SeenEqual = true;
2165 nextToken();
2166 if (FormatTok->is(tok::l_brace)) {
2167 // C# needs this change to ensure that array initialisers and object
2168 // initialisers are indented the same way. In TypeScript, the brace
2169 // can also be an object type definition.
2170 if (!Style.isJavaScript())
2171 FormatTok->setBlockKind(BK_BracedInit);
2172 // TableGen's defset statement has syntax of the form,
2173 // `defset <type> <name> = { <statement>... }`
2174 if (Style.isTableGen() &&
2175 Line->Tokens.begin()->Tok->is(Keywords.kw_defset)) {
2176 FormatTok->setFinalizedType(TT_FunctionLBrace);
2177 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
2178 /*MunchSemi=*/false);
2179 addUnwrappedLine();
2180 break;
2181 }
2182 nextToken();
2183 parseBracedList();
2184 } else if (Style.Language == FormatStyle::LK_Proto &&
2185 FormatTok->is(tok::less)) {
2186 nextToken();
2187 parseBracedList(/*IsAngleBracket=*/true);
2188 }
2189 break;
2190 case tok::l_square:
2191 parseSquare();
2192 break;
2193 case tok::kw_new:
2194 if (Style.isCSharp() &&
2195 (Tokens->peekNextToken()->isAccessSpecifierKeyword() ||
2196 (Previous && Previous->isAccessSpecifierKeyword()))) {
2197 nextToken();
2198 } else {
2199 parseNew();
2200 }
2201 break;
2202 case tok::kw_switch:
2203 if (Style.isJava())
2204 parseSwitch(/*IsExpr=*/true);
2205 else
2206 nextToken();
2207 break;
2208 case tok::kw_case:
2209 // Proto: there are no switch/case statements.
2210 if (Style.Language == FormatStyle::LK_Proto) {
2211 nextToken();
2212 return;
2213 }
2214 // In Verilog switch is called case.
2215 if (Style.isVerilog()) {
2216 parseBlock();
2217 addUnwrappedLine();
2218 return;
2219 }
2220 if (Style.isJavaScript() && Line->MustBeDeclaration) {
2221 // 'case: string' field declaration.
2222 nextToken();
2223 break;
2224 }
2225 parseCaseLabel();
2226 break;
2227 case tok::kw_default:
2228 nextToken();
2229 if (Style.isVerilog()) {
2230 if (FormatTok->is(tok::colon)) {
2231 // The label will be handled in the next iteration.
2232 break;
2233 }
2234 if (FormatTok->is(Keywords.kw_clocking)) {
2235 // A default clocking block.
2236 parseBlock();
2237 addUnwrappedLine();
2238 return;
2239 }
2240 parseVerilogCaseLabel();
2241 return;
2242 }
2243 break;
2244 case tok::colon:
2245 nextToken();
2246 if (Style.isVerilog()) {
2247 parseVerilogCaseLabel();
2248 return;
2249 }
2250 break;
2251 case tok::greater:
2252 nextToken();
2253 if (FormatTok->is(tok::l_brace))
2254 FormatTok->Previous->setFinalizedType(TT_TemplateCloser);
2255 break;
2256 default:
2257 nextToken();
2258 break;
2259 }
2260 }
2261}
2262
2263bool UnwrappedLineParser::tryToParsePropertyAccessor() {
2264 assert(FormatTok->is(tok::l_brace));
2265 if (!Style.isCSharp())
2266 return false;
2267 // See if it's a property accessor.
2268 if (!FormatTok->Previous || FormatTok->Previous->isNot(tok::identifier))
2269 return false;
2270
2271 // See if we are inside a property accessor.
2272 //
2273 // Record the current tokenPosition so that we can advance and
2274 // reset the current token. `Next` is not set yet so we need
2275 // another way to advance along the token stream.
2276 unsigned int StoredPosition = Tokens->getPosition();
2277 FormatToken *Tok = Tokens->getNextToken();
2278
2279 // A trivial property accessor is of the form:
2280 // { [ACCESS_SPECIFIER] [get]; [ACCESS_SPECIFIER] [set|init] }
2281 // Track these as they do not require line breaks to be introduced.
2282 bool HasSpecialAccessor = false;
2283 bool IsTrivialPropertyAccessor = true;
2284 bool HasAttribute = false;
2285 while (!eof()) {
2286 if (const bool IsAccessorKeyword =
2287 Tok->isOneOf(Keywords.kw_get, Keywords.kw_init, Keywords.kw_set);
2288 IsAccessorKeyword || Tok->isAccessSpecifierKeyword() ||
2289 Tok->isOneOf(tok::l_square, tok::semi, Keywords.kw_internal)) {
2290 if (IsAccessorKeyword)
2291 HasSpecialAccessor = true;
2292 else if (Tok->is(tok::l_square))
2293 HasAttribute = true;
2294 Tok = Tokens->getNextToken();
2295 continue;
2296 }
2297 if (Tok->isNot(tok::r_brace))
2298 IsTrivialPropertyAccessor = false;
2299 break;
2300 }
2301
2302 if (!HasSpecialAccessor || HasAttribute) {
2303 Tokens->setPosition(StoredPosition);
2304 return false;
2305 }
2306
2307 // Try to parse the property accessor:
2308 // https://docs.microsoft.com/en-us/dotnet/csharp/programming-guide/classes-and-structs/properties
2309 Tokens->setPosition(StoredPosition);
2310 if (!IsTrivialPropertyAccessor && Style.BraceWrapping.AfterFunction)
2311 addUnwrappedLine();
2312 nextToken();
2313 do {
2314 switch (FormatTok->Tok.getKind()) {
2315 case tok::r_brace:
2316 nextToken();
2317 if (FormatTok->is(tok::equal)) {
2318 while (!eof() && FormatTok->isNot(tok::semi))
2319 nextToken();
2320 nextToken();
2321 }
2322 addUnwrappedLine();
2323 return true;
2324 case tok::l_brace:
2325 ++Line->Level;
2326 parseBlock(/*MustBeDeclaration=*/true);
2327 addUnwrappedLine();
2328 --Line->Level;
2329 break;
2330 case tok::equal:
2331 if (FormatTok->is(TT_FatArrow)) {
2332 ++Line->Level;
2333 do {
2334 nextToken();
2335 } while (!eof() && FormatTok->isNot(tok::semi));
2336 nextToken();
2337 addUnwrappedLine();
2338 --Line->Level;
2339 break;
2340 }
2341 nextToken();
2342 break;
2343 default:
2344 if (FormatTok->isOneOf(Keywords.kw_get, Keywords.kw_init,
2345 Keywords.kw_set) &&
2346 !IsTrivialPropertyAccessor) {
2347 // Non-trivial get/set needs to be on its own line.
2348 addUnwrappedLine();
2349 }
2350 nextToken();
2351 }
2352 } while (!eof());
2353
2354 // Unreachable for well-formed code (paired '{' and '}').
2355 return true;
2356}
2357
2358bool UnwrappedLineParser::tryToParseLambda() {
2359 assert(FormatTok->is(tok::l_square));
2360 if (!IsCpp) {
2361 nextToken();
2362 return false;
2363 }
2364 FormatToken &LSquare = *FormatTok;
2365 if (!tryToParseLambdaIntroducer())
2366 return false;
2367
2368 FormatToken *Arrow = nullptr;
2369 bool InTemplateParameterList = false;
2370
2371 while (FormatTok->isNot(tok::l_brace)) {
2372 if (FormatTok->isTypeName(LangOpts) || FormatTok->isAttribute()) {
2373 nextToken();
2374 continue;
2375 }
2376 switch (FormatTok->Tok.getKind()) {
2377 case tok::l_brace:
2378 break;
2379 case tok::l_paren:
2380 parseParens(/*AmpAmpTokenType=*/TT_PointerOrReference);
2381 break;
2382 case tok::l_square:
2383 parseSquare();
2384 break;
2385 case tok::less:
2386 assert(FormatTok->Previous);
2387 if (FormatTok->Previous->is(tok::r_square))
2388 InTemplateParameterList = true;
2389 nextToken();
2390 break;
2391 case tok::kw_auto:
2392 case tok::kw_class:
2393 case tok::kw_struct:
2394 case tok::kw_union:
2395 case tok::kw_template:
2396 case tok::kw_typename:
2397 case tok::amp:
2398 case tok::star:
2399 case tok::kw_const:
2400 case tok::kw_constexpr:
2401 case tok::kw_consteval:
2402 case tok::comma:
2403 case tok::greater:
2404 case tok::identifier:
2405 case tok::numeric_constant:
2406 case tok::coloncolon:
2407 case tok::kw_mutable:
2408 case tok::kw_noexcept:
2409 case tok::kw_static:
2410 nextToken();
2411 break;
2412 // Specialization of a template with an integer parameter can contain
2413 // arithmetic, logical, comparison and ternary operators.
2414 //
2415 // FIXME: This also accepts sequences of operators that are not in the scope
2416 // of a template argument list.
2417 //
2418 // In a C++ lambda a template type can only occur after an arrow. We use
2419 // this as an heuristic to distinguish between Objective-C expressions
2420 // followed by an `a->b` expression, such as:
2421 // ([obj func:arg] + a->b)
2422 // Otherwise the code below would parse as a lambda.
2423 case tok::plus:
2424 case tok::minus:
2425 case tok::exclaim:
2426 case tok::tilde:
2427 case tok::slash:
2428 case tok::percent:
2429 case tok::lessless:
2430 case tok::pipe:
2431 case tok::pipepipe:
2432 case tok::ampamp:
2433 case tok::caret:
2434 case tok::equalequal:
2435 case tok::exclaimequal:
2436 case tok::greaterequal:
2437 case tok::lessequal:
2438 case tok::question:
2439 case tok::colon:
2440 case tok::ellipsis:
2441 case tok::kw_true:
2442 case tok::kw_false:
2443 if (Arrow || InTemplateParameterList) {
2444 nextToken();
2445 break;
2446 }
2447 return true;
2448 case tok::arrow:
2449 Arrow = FormatTok;
2450 nextToken();
2451 break;
2452 case tok::kw_requires:
2453 parseRequiresClause();
2454 break;
2455 case tok::equal:
2456 if (!InTemplateParameterList)
2457 return true;
2458 nextToken();
2459 break;
2460 default:
2461 return true;
2462 }
2463 }
2464
2465 FormatTok->setFinalizedType(TT_LambdaLBrace);
2466 LSquare.setFinalizedType(TT_LambdaLSquare);
2467
2468 if (Arrow)
2469 Arrow->setFinalizedType(TT_LambdaArrow);
2470
2471 NestedLambdas.push_back(Line->SeenDecltypeAuto);
2472 parseChildBlock();
2473 assert(!NestedLambdas.empty());
2474 NestedLambdas.pop_back();
2475
2476 return true;
2477}
2478
2479bool UnwrappedLineParser::tryToParseLambdaIntroducer() {
2480 const FormatToken *Previous = FormatTok->Previous;
2481 const FormatToken *LeftSquare = FormatTok;
2482 nextToken();
2483 if (Previous) {
2484 const auto *PrevPrev = Previous->getPreviousNonComment();
2485 if (Previous->is(tok::star) && PrevPrev && PrevPrev->isTypeName(LangOpts))
2486 return false;
2487 if (Previous->closesScope()) {
2488 // Not a potential C-style cast.
2489 if (Previous->isNot(tok::r_paren))
2490 return false;
2491 // Lambdas can be cast to function types only, e.g. `std::function<int()>`
2492 // and `int (*)()`.
2493 if (!PrevPrev || PrevPrev->isNoneOf(tok::greater, tok::r_paren))
2494 return false;
2495 }
2496 if (Previous && Previous->Tok.getIdentifierInfo() &&
2497 Previous->isNoneOf(tok::kw_return, tok::kw_co_await, tok::kw_co_yield,
2498 tok::kw_co_return)) {
2499 return false;
2500 }
2501 }
2502 if (LeftSquare->isCppStructuredBinding(IsCpp))
2503 return false;
2504 if (FormatTok->is(tok::l_square) || tok::isLiteral(FormatTok->Tok.getKind()))
2505 return false;
2506 if (FormatTok->is(tok::r_square)) {
2507 const FormatToken *Next = Tokens->peekNextToken(/*SkipComment=*/true);
2508 if (Next->is(tok::greater))
2509 return false;
2510 }
2511 parseSquare(/*LambdaIntroducer=*/true);
2512 return true;
2513}
2514
2515void UnwrappedLineParser::tryToParseJSFunction() {
2516 assert(FormatTok->is(Keywords.kw_function));
2517 if (FormatTok->is(Keywords.kw_async))
2518 nextToken();
2519 // Consume "function".
2520 nextToken();
2521
2522 // Consume * (generator function). Treat it like C++'s overloaded operators.
2523 if (FormatTok->is(tok::star)) {
2524 FormatTok->setFinalizedType(TT_OverloadedOperator);
2525 nextToken();
2526 }
2527
2528 // Consume function name.
2529 if (FormatTok->is(tok::identifier))
2530 nextToken();
2531
2532 if (FormatTok->isNot(tok::l_paren))
2533 return;
2534
2535 // Parse formal parameter list.
2536 parseParens();
2537
2538 if (FormatTok->is(tok::colon)) {
2539 // Parse a type definition.
2540 nextToken();
2541
2542 // Eat the type declaration. For braced inline object types, balance braces,
2543 // otherwise just parse until finding an l_brace for the function body.
2544 if (FormatTok->is(tok::l_brace))
2545 tryToParseBracedList();
2546 else
2547 while (FormatTok->isNoneOf(tok::l_brace, tok::semi) && !eof())
2548 nextToken();
2549 }
2550
2551 if (FormatTok->is(tok::semi))
2552 return;
2553
2554 parseChildBlock();
2555}
2556
2557bool UnwrappedLineParser::tryToParseBracedList() {
2558 if (FormatTok->is(BK_Unknown))
2559 calculateBraceTypes();
2560 assert(FormatTok->isNot(BK_Unknown));
2561 if (FormatTok->is(BK_Block))
2562 return false;
2563 nextToken();
2564 parseBracedList();
2565 return true;
2566}
2567
2568bool UnwrappedLineParser::tryToParseChildBlock() {
2569 assert(Style.isJavaScript() || Style.isCSharp());
2570 assert(FormatTok->is(TT_FatArrow));
2571 // Fat arrows (=>) have tok::TokenKind tok::equal but TokenType TT_FatArrow.
2572 // They always start an expression or a child block if followed by a curly
2573 // brace.
2574 nextToken();
2575 if (FormatTok->isNot(tok::l_brace))
2576 return false;
2577 parseChildBlock();
2578 return true;
2579}
2580
2581bool UnwrappedLineParser::parseBracedList(bool IsAngleBracket, bool IsEnum) {
2582 assert(!IsAngleBracket || !IsEnum);
2583 bool HasError = false;
2584
2585 // FIXME: Once we have an expression parser in the UnwrappedLineParser,
2586 // replace this by using parseAssignmentExpression() inside.
2587 do {
2588 if (Style.isCSharp() && FormatTok->is(TT_FatArrow) &&
2589 tryToParseChildBlock()) {
2590 continue;
2591 }
2592 if (Style.isJavaScript()) {
2593 if (FormatTok->is(Keywords.kw_function)) {
2594 tryToParseJSFunction();
2595 continue;
2596 }
2597 if (FormatTok->is(tok::l_brace)) {
2598 // Could be a method inside of a braced list `{a() { return 1; }}`.
2599 if (tryToParseBracedList())
2600 continue;
2601 parseChildBlock();
2602 }
2603 }
2604 if (FormatTok->is(IsAngleBracket ? tok::greater : tok::r_brace)) {
2605 if (IsEnum) {
2606 FormatTok->setBlockKind(BK_Block);
2607 if (!Style.AllowShortEnumsOnASingleLine)
2608 addUnwrappedLine();
2609 }
2610 nextToken();
2611 return !HasError;
2612 }
2613 switch (FormatTok->Tok.getKind()) {
2614 case tok::l_square:
2615 if (Style.isCSharp())
2616 parseSquare();
2617 else
2618 tryToParseLambda();
2619 break;
2620 case tok::l_paren:
2621 parseParens();
2622 // JavaScript can just have free standing methods and getters/setters in
2623 // object literals. Detect them by a "{" following ")".
2624 if (Style.isJavaScript()) {
2625 if (FormatTok->is(tok::l_brace))
2626 parseChildBlock();
2627 break;
2628 }
2629 break;
2630 case tok::l_brace:
2631 // Assume there are no blocks inside a braced init list apart
2632 // from the ones we explicitly parse out (like lambdas).
2633 FormatTok->setBlockKind(BK_BracedInit);
2634 if (!IsAngleBracket) {
2635 auto *Prev = FormatTok->Previous;
2636 if (Prev && Prev->is(tok::greater))
2637 Prev->setFinalizedType(TT_TemplateCloser);
2638 }
2639 nextToken();
2640 parseBracedList();
2641 break;
2642 case tok::less:
2643 nextToken();
2644 if (IsAngleBracket)
2645 parseBracedList(/*IsAngleBracket=*/true);
2646 break;
2647 case tok::semi:
2648 // JavaScript (or more precisely TypeScript) can have semicolons in braced
2649 // lists (in so-called TypeMemberLists). Thus, the semicolon cannot be
2650 // used for error recovery if we have otherwise determined that this is
2651 // a braced list.
2652 if (Style.isJavaScript()) {
2653 nextToken();
2654 break;
2655 }
2656 HasError = true;
2657 if (!IsEnum)
2658 return false;
2659 nextToken();
2660 break;
2661 case tok::comma:
2662 nextToken();
2663 if (IsEnum && !Style.AllowShortEnumsOnASingleLine)
2664 addUnwrappedLine();
2665 break;
2666 case tok::kw_requires:
2667 parseRequiresExpression();
2668 break;
2669 default:
2670 nextToken();
2671 break;
2672 }
2673 } while (!eof());
2674 return false;
2675}
2676
2677/// Parses a pair of parentheses (and everything between them).
2678/// \param StarAndAmpTokenType If different than TT_Unknown sets this type for
2679/// all (double) ampersands and stars. This applies for all nested scopes as
2680/// well, this is disabled within a (potential) template argument <>, and thus
2681/// also if we find only a <.
2682///
2683/// Returns whether there is a `=` token between the parentheses.
2684bool UnwrappedLineParser::parseParens(TokenType StarAndAmpTokenType,
2685 bool InMacroCall) {
2686 assert(FormatTok->is(tok::l_paren) && "'(' expected.");
2687 auto *LParen = FormatTok;
2688 auto *Prev = FormatTok->Previous;
2689 bool SeenComma = false;
2690 bool SeenEqual = false;
2691 bool MightBeFoldExpr = false;
2692 unsigned ExcessLess = 0;
2693 nextToken();
2694 const bool MightBeStmtExpr = FormatTok->is(tok::l_brace);
2695 if (!InMacroCall && Prev && Prev->is(TT_FunctionLikeMacro))
2696 InMacroCall = true;
2697 do {
2698 switch (FormatTok->Tok.getKind()) {
2699 case tok::l_paren:
2700 if (parseParens(ExcessLess == 0 ? StarAndAmpTokenType : TT_Unknown,
2701 InMacroCall)) {
2702 SeenEqual = true;
2703 }
2704 if (Style.isJava() && FormatTok->is(tok::l_brace))
2705 parseChildBlock();
2706 break;
2707 case tok::r_paren: {
2708 auto *RParen = FormatTok;
2709 nextToken();
2710 if (Prev) {
2711 auto OptionalParens = [&] {
2712 if (Style.RemoveParentheses == FormatStyle::RPS_Leave ||
2713 MightBeStmtExpr || MightBeFoldExpr || SeenComma || InMacroCall ||
2714 Line->InMacroBody || RParen->getPreviousNonComment() == LParen) {
2715 return false;
2716 }
2717 const bool DoubleParens =
2718 Prev->is(tok::l_paren) && FormatTok->is(tok::r_paren);
2719 if (DoubleParens) {
2720 const auto *PrevPrev = Prev->getPreviousNonComment();
2721 const bool Excluded =
2722 PrevPrev &&
2723 (PrevPrev->isOneOf(tok::kw___attribute, tok::kw_decltype) ||
2724 (SeenEqual &&
2725 (PrevPrev->isOneOf(tok::kw_if, tok::kw_while) ||
2726 PrevPrev->endsSequence(tok::kw_constexpr, tok::kw_if))));
2727 if (!Excluded)
2728 return true;
2729 } else {
2730 const bool CommaSeparated =
2731 Prev->isOneOf(tok::l_paren, tok::comma) &&
2732 FormatTok->isOneOf(tok::comma, tok::r_paren);
2733 if (CommaSeparated &&
2734 // LParen is not preceded by ellipsis, comma.
2735 !Prev->endsSequence(tok::comma, tok::ellipsis) &&
2736 // RParen is not followed by comma, ellipsis.
2737 !(FormatTok->is(tok::comma) &&
2738 Tokens->peekNextToken()->is(tok::ellipsis))) {
2739 return true;
2740 }
2741 const bool ReturnParens =
2742 Style.RemoveParentheses == FormatStyle::RPS_ReturnStatement &&
2743 ((NestedLambdas.empty() && !IsDecltypeAutoFunction) ||
2744 (!NestedLambdas.empty() && !NestedLambdas.back())) &&
2745 Prev->isOneOf(tok::kw_return, tok::kw_co_return) &&
2746 FormatTok->is(tok::semi);
2747 if (ReturnParens)
2748 return true;
2749 }
2750 return false;
2751 };
2752 if (OptionalParens()) {
2753 LParen->Optional = true;
2754 RParen->Optional = true;
2755 } else if (Prev->is(TT_TypenameMacro)) {
2756 LParen->setFinalizedType(TT_TypeDeclarationParen);
2757 RParen->setFinalizedType(TT_TypeDeclarationParen);
2758 } else if (Prev->is(tok::greater) && RParen->Previous == LParen) {
2759 Prev->setFinalizedType(TT_TemplateCloser);
2760 } else if (FormatTok->is(tok::l_brace) && Prev->is(tok::amp) &&
2761 !Prev->Previous) {
2762 FormatTok->setBlockKind(BK_BracedInit);
2763 }
2764 }
2765 return SeenEqual;
2766 }
2767 case tok::r_brace:
2768 // A "}" inside parenthesis is an error if there wasn't a matching "{".
2769 return SeenEqual;
2770 case tok::l_square:
2771 tryToParseLambda();
2772 break;
2773 case tok::l_brace:
2774 if (!tryToParseBracedList())
2775 parseChildBlock();
2776 break;
2777 case tok::at:
2778 nextToken();
2779 if (FormatTok->is(tok::l_brace)) {
2780 nextToken();
2781 parseBracedList();
2782 }
2783 break;
2784 case tok::comma:
2785 SeenComma = true;
2786 nextToken();
2787 break;
2788 case tok::ellipsis:
2789 MightBeFoldExpr = true;
2790 nextToken();
2791 break;
2792 case tok::equal:
2793 SeenEqual = true;
2794 if (Style.isCSharp() && FormatTok->is(TT_FatArrow))
2795 tryToParseChildBlock();
2796 else
2797 nextToken();
2798 break;
2799 case tok::kw_class:
2800 if (Style.isJavaScript())
2801 parseRecord(/*ParseAsExpr=*/true);
2802 else
2803 nextToken();
2804 break;
2805 case tok::identifier:
2806 if (Style.isJavaScript() && (FormatTok->is(Keywords.kw_function)))
2807 tryToParseJSFunction();
2808 else
2809 nextToken();
2810 break;
2811 case tok::kw_switch:
2812 if (Style.isJava())
2813 parseSwitch(/*IsExpr=*/true);
2814 else
2815 nextToken();
2816 break;
2817 case tok::kw_requires:
2818 parseRequiresExpression();
2819 break;
2820 case tok::less:
2821 // We have here no clue whether this is a less, or a template opener, opt
2822 // out of the predefined StarAndAmpTokenType.
2823 ++ExcessLess;
2824 nextToken();
2825 break;
2826 case tok::greater:
2827 if (ExcessLess > 0)
2828 --ExcessLess;
2829 nextToken();
2830 break;
2831 case tok::star:
2832 case tok::amp:
2833 case tok::ampamp:
2834 if (StarAndAmpTokenType != TT_Unknown && ExcessLess == 0)
2835 FormatTok->setFinalizedType(StarAndAmpTokenType);
2836 [[fallthrough]];
2837 default:
2838 nextToken();
2839 break;
2840 }
2841 } while (!eof());
2842 return SeenEqual;
2843}
2844
2845void UnwrappedLineParser::parseSquare(bool LambdaIntroducer) {
2846 if (!LambdaIntroducer) {
2847 assert(FormatTok->is(tok::l_square) && "'[' expected.");
2848 if (tryToParseLambda())
2849 return;
2850 }
2851 do {
2852 switch (FormatTok->Tok.getKind()) {
2853 case tok::l_paren:
2854 parseParens();
2855 break;
2856 case tok::r_square:
2857 nextToken();
2858 return;
2859 case tok::r_brace:
2860 // A "}" inside parenthesis is an error if there wasn't a matching "{".
2861 return;
2862 case tok::l_square:
2863 parseSquare();
2864 break;
2865 case tok::l_brace: {
2866 if (!tryToParseBracedList())
2867 parseChildBlock();
2868 break;
2869 }
2870 case tok::at:
2871 case tok::colon:
2872 nextToken();
2873 if (FormatTok->is(tok::l_brace)) {
2874 nextToken();
2875 parseBracedList();
2876 }
2877 break;
2878 default:
2879 nextToken();
2880 break;
2881 }
2882 } while (!eof());
2883}
2884
2885void UnwrappedLineParser::keepAncestorBraces() {
2886 if (!Style.RemoveBracesLLVM)
2887 return;
2888
2889 const int MaxNestingLevels = 2;
2890 const int Size = NestedTooDeep.size();
2891 if (Size >= MaxNestingLevels)
2892 NestedTooDeep[Size - MaxNestingLevels] = true;
2893 NestedTooDeep.push_back(false);
2894}
2895
2897 for (const auto &Token : llvm::reverse(Line.Tokens))
2898 if (Token.Tok->isNot(tok::comment))
2899 return Token.Tok;
2900
2901 return nullptr;
2902}
2903
2904void UnwrappedLineParser::parseUnbracedBody(bool CheckEOF) {
2905 FormatToken *Tok = nullptr;
2906
2907 if (Style.InsertBraces && !Line->InPPDirective && !Line->Tokens.empty() &&
2908 PreprocessorDirectives.empty() && FormatTok->isNot(tok::semi)) {
2909 Tok = Style.BraceWrapping.AfterControlStatement == FormatStyle::BWACS_Never
2910 ? getLastNonComment(*Line)
2911 : Line->Tokens.back().Tok;
2912 assert(Tok);
2913 if (Tok->BraceCount < 0) {
2914 assert(Tok->BraceCount == -1);
2915 Tok = nullptr;
2916 } else {
2917 Tok->BraceCount = -1;
2918 }
2919 }
2920
2921 addUnwrappedLine();
2922 ++Line->Level;
2923 ++Line->UnbracedBodyLevel;
2924 parseStructuralElement();
2925 --Line->UnbracedBodyLevel;
2926
2927 if (Tok) {
2928 assert(!Line->InPPDirective);
2929 Tok = nullptr;
2930 for (const auto &L : llvm::reverse(*CurrentLines)) {
2931 if (!L.InPPDirective && getLastNonComment(L)) {
2932 Tok = L.Tokens.back().Tok;
2933 break;
2934 }
2935 }
2936 assert(Tok);
2937 ++Tok->BraceCount;
2938 }
2939
2940 if (CheckEOF && eof())
2941 addUnwrappedLine();
2942
2943 --Line->Level;
2944}
2945
2946static void markOptionalBraces(FormatToken *LeftBrace) {
2947 if (!LeftBrace)
2948 return;
2949
2950 assert(LeftBrace->is(tok::l_brace));
2951
2952 FormatToken *RightBrace = LeftBrace->MatchingParen;
2953 if (!RightBrace) {
2954 assert(!LeftBrace->Optional);
2955 return;
2956 }
2957
2958 assert(RightBrace->is(tok::r_brace));
2959 assert(RightBrace->MatchingParen == LeftBrace);
2960 assert(LeftBrace->Optional == RightBrace->Optional);
2961
2962 LeftBrace->Optional = true;
2963 RightBrace->Optional = true;
2964}
2965
2966void UnwrappedLineParser::handleAttributes() {
2967 // Handle AttributeMacro, e.g. `if (x) UNLIKELY`.
2968 if (FormatTok->isAttribute())
2969 nextToken();
2970 else if (FormatTok->is(tok::l_square))
2971 handleCppAttributes();
2972}
2973
2974bool UnwrappedLineParser::handleCppAttributes() {
2975 // Handle [[likely]] / [[unlikely]] attributes.
2976 assert(FormatTok->is(tok::l_square));
2977 if (!tryToParseSimpleAttribute())
2978 return false;
2979 parseSquare();
2980 return true;
2981}
2982
2983/// Returns whether \c Tok begins a block.
2984bool UnwrappedLineParser::isBlockBegin(const FormatToken &Tok) const {
2985 // FIXME: rename the function or make
2986 // Tok.isOneOf(tok::l_brace, TT_MacroBlockBegin) work.
2987 return Style.isVerilog() ? Keywords.isVerilogBegin(Tok)
2988 : Tok.is(tok::l_brace);
2989}
2990
2991FormatToken *UnwrappedLineParser::parseIfThenElse(IfStmtKind *IfKind,
2992 bool KeepBraces,
2993 bool IsVerilogAssert) {
2994 assert((FormatTok->is(tok::kw_if) ||
2995 (Style.isVerilog() &&
2996 FormatTok->isOneOf(tok::kw_restrict, Keywords.kw_assert,
2997 Keywords.kw_assume, Keywords.kw_cover))) &&
2998 "'if' expected");
2999 nextToken();
3000
3001 if (IsVerilogAssert) {
3002 // Handle `assert #0` and `assert final`.
3003 if (FormatTok->is(Keywords.kw_verilogHash)) {
3004 nextToken();
3005 if (FormatTok->is(tok::numeric_constant))
3006 nextToken();
3007 } else if (FormatTok->isOneOf(Keywords.kw_final, Keywords.kw_property,
3008 Keywords.kw_sequence)) {
3009 nextToken();
3010 }
3011 }
3012
3013 // TableGen's if statement has the form of `if <cond> then { ... }`.
3014 if (Style.isTableGen()) {
3015 while (!eof() && FormatTok->isNot(Keywords.kw_then)) {
3016 // Simply skip until then. This range only contains a value.
3017 nextToken();
3018 }
3019 }
3020
3021 // Handle `if !consteval`.
3022 if (FormatTok->is(tok::exclaim))
3023 nextToken();
3024
3025 bool KeepIfBraces = true;
3026 if (FormatTok->is(tok::kw_consteval)) {
3027 nextToken();
3028 } else {
3029 KeepIfBraces = !Style.RemoveBracesLLVM || KeepBraces;
3030 if (FormatTok->isOneOf(tok::kw_constexpr, tok::identifier))
3031 nextToken();
3032 if (FormatTok->is(tok::l_paren)) {
3033 FormatTok->setFinalizedType(TT_ConditionLParen);
3034 parseParens();
3035 }
3036 }
3037 handleAttributes();
3038 // The then action is optional in Verilog assert statements.
3039 if (IsVerilogAssert && FormatTok->is(tok::semi)) {
3040 nextToken();
3041 addUnwrappedLine();
3042 return nullptr;
3043 }
3044
3045 bool NeedsUnwrappedLine = false;
3046 keepAncestorBraces();
3047
3048 FormatToken *IfLeftBrace = nullptr;
3049 IfStmtKind IfBlockKind = IfStmtKind::NotIf;
3050
3051 if (isBlockBegin(*FormatTok)) {
3052 FormatTok->setFinalizedType(TT_ControlStatementLBrace);
3053 IfLeftBrace = FormatTok;
3054 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3055 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
3056 /*MunchSemi=*/true, KeepIfBraces, &IfBlockKind);
3057 setPreviousRBraceType(TT_ControlStatementRBrace);
3058 if (Style.BraceWrapping.BeforeElse)
3059 addUnwrappedLine();
3060 else
3061 NeedsUnwrappedLine = true;
3062 } else if (IsVerilogAssert && FormatTok->is(tok::kw_else)) {
3063 addUnwrappedLine();
3064 } else {
3065 parseUnbracedBody();
3066 }
3067
3068 if (Style.RemoveBracesLLVM) {
3069 assert(!NestedTooDeep.empty());
3070 KeepIfBraces = KeepIfBraces ||
3071 (IfLeftBrace && !IfLeftBrace->MatchingParen) ||
3072 NestedTooDeep.back() || IfBlockKind == IfStmtKind::IfOnly ||
3073 IfBlockKind == IfStmtKind::IfElseIf;
3074 }
3075
3076 bool KeepElseBraces = KeepIfBraces;
3077 FormatToken *ElseLeftBrace = nullptr;
3078 IfStmtKind Kind = IfStmtKind::IfOnly;
3079
3080 if (FormatTok->is(tok::kw_else)) {
3081 if (Style.RemoveBracesLLVM) {
3082 NestedTooDeep.back() = false;
3083 Kind = IfStmtKind::IfElse;
3084 }
3085 nextToken();
3086 handleAttributes();
3087 if (isBlockBegin(*FormatTok)) {
3088 const bool FollowedByIf = Tokens->peekNextToken()->is(tok::kw_if);
3089 FormatTok->setFinalizedType(TT_ElseLBrace);
3090 ElseLeftBrace = FormatTok;
3091 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3092 IfStmtKind ElseBlockKind = IfStmtKind::NotIf;
3093 FormatToken *IfLBrace =
3094 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
3095 /*MunchSemi=*/true, KeepElseBraces, &ElseBlockKind);
3096 setPreviousRBraceType(TT_ElseRBrace);
3097 if (FormatTok->is(tok::kw_else)) {
3098 KeepElseBraces = KeepElseBraces ||
3099 ElseBlockKind == IfStmtKind::IfOnly ||
3100 ElseBlockKind == IfStmtKind::IfElseIf;
3101 } else if (FollowedByIf && IfLBrace && !IfLBrace->Optional) {
3102 KeepElseBraces = true;
3103 assert(ElseLeftBrace->MatchingParen);
3104 markOptionalBraces(ElseLeftBrace);
3105 }
3106 addUnwrappedLine();
3107 } else if (!IsVerilogAssert && FormatTok->is(tok::kw_if)) {
3108 const FormatToken *Previous = Tokens->getPreviousToken();
3109 assert(Previous);
3110 const bool IsPrecededByComment = Previous->is(tok::comment);
3111 if (IsPrecededByComment) {
3112 addUnwrappedLine();
3113 ++Line->Level;
3114 }
3115 bool TooDeep = true;
3116 if (Style.RemoveBracesLLVM) {
3117 Kind = IfStmtKind::IfElseIf;
3118 TooDeep = NestedTooDeep.pop_back_val();
3119 }
3120 ElseLeftBrace = parseIfThenElse(/*IfKind=*/nullptr, KeepIfBraces);
3121 if (Style.RemoveBracesLLVM)
3122 NestedTooDeep.push_back(TooDeep);
3123 if (IsPrecededByComment)
3124 --Line->Level;
3125 } else {
3126 parseUnbracedBody(/*CheckEOF=*/true);
3127 }
3128 } else {
3129 KeepIfBraces = KeepIfBraces || IfBlockKind == IfStmtKind::IfElse;
3130 if (NeedsUnwrappedLine)
3131 addUnwrappedLine();
3132 }
3133
3134 if (!Style.RemoveBracesLLVM)
3135 return nullptr;
3136
3137 assert(!NestedTooDeep.empty());
3138 KeepElseBraces = KeepElseBraces ||
3139 (ElseLeftBrace && !ElseLeftBrace->MatchingParen) ||
3140 NestedTooDeep.back();
3141
3142 NestedTooDeep.pop_back();
3143
3144 if (!KeepIfBraces && !KeepElseBraces) {
3145 markOptionalBraces(IfLeftBrace);
3146 markOptionalBraces(ElseLeftBrace);
3147 } else if (IfLeftBrace) {
3148 FormatToken *IfRightBrace = IfLeftBrace->MatchingParen;
3149 if (IfRightBrace) {
3150 assert(IfRightBrace->MatchingParen == IfLeftBrace);
3151 assert(!IfLeftBrace->Optional);
3152 assert(!IfRightBrace->Optional);
3153 IfLeftBrace->MatchingParen = nullptr;
3154 IfRightBrace->MatchingParen = nullptr;
3155 }
3156 }
3157
3158 if (IfKind)
3159 *IfKind = Kind;
3160
3161 return IfLeftBrace;
3162}
3163
3164void UnwrappedLineParser::parseTryCatch() {
3165 assert(FormatTok->isOneOf(tok::kw_try, tok::kw___try) && "'try' expected");
3166 nextToken();
3167 bool NeedsUnwrappedLine = false;
3168 bool HasCtorInitializer = false;
3169 if (FormatTok->is(tok::colon)) {
3170 auto *Colon = FormatTok;
3171 // We are in a function try block, what comes is an initializer list.
3172 nextToken();
3173 if (FormatTok->is(tok::identifier)) {
3174 HasCtorInitializer = true;
3175 Colon->setFinalizedType(TT_CtorInitializerColon);
3176 }
3177
3178 // In case identifiers were removed by clang-tidy, what might follow is
3179 // multiple commas in sequence - before the first identifier.
3180 while (FormatTok->is(tok::comma))
3181 nextToken();
3182
3183 while (FormatTok->is(tok::identifier)) {
3184 nextToken();
3185 if (FormatTok->is(tok::l_paren)) {
3186 parseParens();
3187 } else if (FormatTok->is(tok::l_brace)) {
3188 nextToken();
3189 parseBracedList();
3190 }
3191
3192 // In case identifiers were removed by clang-tidy, what might follow is
3193 // multiple commas in sequence - after the first identifier.
3194 while (FormatTok->is(tok::comma))
3195 nextToken();
3196 }
3197 }
3198 // Parse try with resource.
3199 if (Style.isJava() && FormatTok->is(tok::l_paren))
3200 parseParens();
3201
3202 keepAncestorBraces();
3203
3204 if (FormatTok->is(tok::l_brace)) {
3205 if (HasCtorInitializer)
3206 FormatTok->setFinalizedType(TT_FunctionLBrace);
3207 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3208 parseBlock();
3209 if (Style.BraceWrapping.BeforeCatch)
3210 addUnwrappedLine();
3211 else
3212 NeedsUnwrappedLine = true;
3213 } else if (FormatTok->isNot(tok::kw_catch)) {
3214 // The C++ standard requires a compound-statement after a try.
3215 // If there's none, we try to assume there's a structuralElement
3216 // and try to continue.
3217 addUnwrappedLine();
3218 ++Line->Level;
3219 parseStructuralElement();
3220 --Line->Level;
3221 }
3222 for (bool SeenCatch = false;;) {
3223 if (FormatTok->is(tok::at))
3224 nextToken();
3225 if (FormatTok->isNoneOf(tok::kw_catch, Keywords.kw___except,
3226 tok::kw___finally, tok::objc_catch,
3227 tok::objc_finally) &&
3228 !((Style.isJava() || Style.isJavaScript()) &&
3229 FormatTok->is(Keywords.kw_finally))) {
3230 break;
3231 }
3232 if (FormatTok->is(tok::kw_catch))
3233 SeenCatch = true;
3234 nextToken();
3235 while (FormatTok->isNot(tok::l_brace)) {
3236 if (FormatTok->is(tok::l_paren)) {
3237 parseParens();
3238 continue;
3239 }
3240 if (FormatTok->isOneOf(tok::semi, tok::r_brace) || eof()) {
3241 if (Style.RemoveBracesLLVM)
3242 NestedTooDeep.pop_back();
3243 return;
3244 }
3245 nextToken();
3246 }
3247 if (SeenCatch) {
3248 FormatTok->setFinalizedType(TT_ControlStatementLBrace);
3249 SeenCatch = false;
3250 }
3251 NeedsUnwrappedLine = false;
3252 Line->MustBeDeclaration = false;
3253 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3254 parseBlock();
3255 if (Style.BraceWrapping.BeforeCatch)
3256 addUnwrappedLine();
3257 else
3258 NeedsUnwrappedLine = true;
3259 }
3260
3261 if (Style.RemoveBracesLLVM)
3262 NestedTooDeep.pop_back();
3263
3264 if (NeedsUnwrappedLine)
3265 addUnwrappedLine();
3266}
3267
3268void UnwrappedLineParser::parseNamespaceOrExportBlock(unsigned AddLevels) {
3269 bool ManageWhitesmithsBraces =
3270 AddLevels == 0u && Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths;
3271
3272 // If we're in Whitesmiths mode, indent the brace if we're not indenting
3273 // the whole block.
3274 if (ManageWhitesmithsBraces)
3275 ++Line->Level;
3276
3277 // Munch the semicolon after the block. This is more common than one would
3278 // think. Putting the semicolon into its own line is very ugly.
3279 parseBlock(/*MustBeDeclaration=*/true, AddLevels, /*MunchSemi=*/true,
3280 /*KeepBraces=*/true, /*IfKind=*/nullptr, ManageWhitesmithsBraces);
3281
3282 addUnwrappedLine(AddLevels > 0 ? LineLevel::Remove : LineLevel::Keep);
3283
3284 if (ManageWhitesmithsBraces)
3285 --Line->Level;
3286}
3287
3288void UnwrappedLineParser::parseNamespace() {
3289 assert(FormatTok->isOneOf(tok::kw_namespace, TT_NamespaceMacro) &&
3290 "'namespace' expected");
3291
3292 const FormatToken &InitialToken = *FormatTok;
3293 nextToken();
3294 if (InitialToken.is(TT_NamespaceMacro)) {
3295 parseParens();
3296 } else {
3297 while (FormatTok->isOneOf(tok::identifier, tok::coloncolon, tok::kw_inline,
3298 tok::l_square, tok::period, tok::l_paren) ||
3299 (Style.isCSharp() && FormatTok->is(tok::kw_union))) {
3300 if (FormatTok->is(tok::l_square))
3301 parseSquare();
3302 else if (FormatTok->is(tok::l_paren))
3303 parseParens();
3304 else
3305 nextToken();
3306 }
3307 }
3308 if (FormatTok->is(tok::l_brace)) {
3309 FormatTok->setFinalizedType(TT_NamespaceLBrace);
3310
3311 if (ShouldBreakBeforeBrace(Style, InitialToken,
3312 Tokens->peekNextToken()->is(tok::r_brace))) {
3313 addUnwrappedLine();
3314 }
3315
3316 unsigned AddLevels =
3317 Style.NamespaceIndentation == FormatStyle::NI_All ||
3318 (Style.NamespaceIndentation == FormatStyle::NI_Inner &&
3319 DeclarationScopeStack.size() > 1)
3320 ? 1u
3321 : 0u;
3322 parseNamespaceOrExportBlock(AddLevels);
3323 }
3324 // FIXME: Add error handling.
3325}
3326
3327void UnwrappedLineParser::parseCppExportBlock() {
3328 if (FormatTok->is(tok::l_brace)) {
3329 FormatTok->setFinalizedType(TT_ExportLBrace);
3330 if (Style.BraceWrapping.AfterExportBlock)
3331 addUnwrappedLine();
3332 }
3333 parseNamespaceOrExportBlock(/*AddLevels=*/Style.IndentExportBlock ? 1 : 0);
3334}
3335
3336void UnwrappedLineParser::parseNew() {
3337 assert(FormatTok->is(tok::kw_new) && "'new' expected");
3338 nextToken();
3339
3340 if (Style.isCSharp()) {
3341 do {
3342 // Handle constructor invocation, e.g. `new(field: value)`.
3343 if (FormatTok->is(tok::l_paren))
3344 parseParens();
3345
3346 // Handle array initialization syntax, e.g. `new[] {10, 20, 30}`.
3347 if (FormatTok->is(tok::l_brace))
3348 parseBracedList();
3349
3350 if (FormatTok->isOneOf(tok::semi, tok::comma))
3351 return;
3352
3353 nextToken();
3354 } while (!eof());
3355 }
3356
3357 if (!Style.isJava())
3358 return;
3359
3360 // In Java, we can parse everything up to the parens, which aren't optional.
3361 do {
3362 // There should not be a ;, { or } before the new's open paren.
3363 if (FormatTok->isOneOf(tok::semi, tok::l_brace, tok::r_brace))
3364 return;
3365
3366 // Consume the parens.
3367 if (FormatTok->is(tok::l_paren)) {
3368 parseParens();
3369
3370 // If there is a class body of an anonymous class, consume that as child.
3371 if (FormatTok->is(tok::l_brace))
3372 parseChildBlock();
3373 return;
3374 }
3375 nextToken();
3376 } while (!eof());
3377}
3378
3379void UnwrappedLineParser::parseLoopBody(bool KeepBraces, bool WrapRightBrace) {
3380 keepAncestorBraces();
3381
3382 if (isBlockBegin(*FormatTok)) {
3383 FormatTok->setFinalizedType(TT_ControlStatementLBrace);
3384 FormatToken *LeftBrace = FormatTok;
3385 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3386 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
3387 /*MunchSemi=*/true, KeepBraces);
3388 setPreviousRBraceType(TT_ControlStatementRBrace);
3389 if (!KeepBraces) {
3390 assert(!NestedTooDeep.empty());
3391 if (!NestedTooDeep.back())
3392 markOptionalBraces(LeftBrace);
3393 }
3394 if (WrapRightBrace)
3395 addUnwrappedLine();
3396 } else {
3397 parseUnbracedBody();
3398 }
3399
3400 if (!KeepBraces)
3401 NestedTooDeep.pop_back();
3402}
3403
3404void UnwrappedLineParser::parseForOrWhileLoop(bool HasParens) {
3405 assert((FormatTok->isOneOf(tok::kw_for, tok::kw_while, TT_ForEachMacro) ||
3406 (Style.isVerilog() &&
3407 FormatTok->isOneOf(Keywords.kw_always, Keywords.kw_always_comb,
3408 Keywords.kw_always_ff, Keywords.kw_always_latch,
3409 Keywords.kw_final, Keywords.kw_initial,
3410 Keywords.kw_foreach, Keywords.kw_forever,
3411 Keywords.kw_repeat))) &&
3412 "'for', 'while' or foreach macro expected");
3413 const bool KeepBraces = !Style.RemoveBracesLLVM ||
3414 FormatTok->isNoneOf(tok::kw_for, tok::kw_while);
3415
3416 nextToken();
3417 // JS' for await ( ...
3418 if (Style.isJavaScript() && FormatTok->is(Keywords.kw_await))
3419 nextToken();
3420 if (IsCpp && FormatTok->is(tok::kw_co_await))
3421 nextToken();
3422 if (HasParens && FormatTok->is(tok::l_paren)) {
3423 // The type is only set for Verilog basically because we were afraid to
3424 // change the existing behavior for loops. See the discussion on D121756 for
3425 // details.
3426 if (Style.isVerilog())
3427 FormatTok->setFinalizedType(TT_ConditionLParen);
3428 parseParens();
3429 }
3430
3431 if (Style.isVerilog()) {
3432 // Event control.
3433 parseVerilogSensitivityList();
3434 } else if (Style.AllowShortLoopsOnASingleLine && FormatTok->is(tok::semi) &&
3435 Tokens->getPreviousToken()->is(tok::r_paren)) {
3436 nextToken();
3437 addUnwrappedLine();
3438 return;
3439 }
3440
3441 handleAttributes();
3442 parseLoopBody(KeepBraces, /*WrapRightBrace=*/true);
3443}
3444
3445void UnwrappedLineParser::parseDoWhile() {
3446 assert(FormatTok->is(tok::kw_do) && "'do' expected");
3447 nextToken();
3448
3449 parseLoopBody(/*KeepBraces=*/true, Style.BraceWrapping.BeforeWhile);
3450
3451 // FIXME: Add error handling.
3452 if (FormatTok->isNot(tok::kw_while)) {
3453 addUnwrappedLine();
3454 return;
3455 }
3456
3457 FormatTok->setFinalizedType(TT_DoWhile);
3458
3459 // If in Whitesmiths mode, the line with the while() needs to be indented
3460 // to the same level as the block.
3461 if (Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths)
3462 ++Line->Level;
3463
3464 nextToken();
3465 parseStructuralElement();
3466}
3467
3468void UnwrappedLineParser::parseLabel(bool IsGotoLabel) {
3469 nextToken();
3470
3471 const auto IndentGotoLabel = Style.IndentGotoLabels;
3472 const auto OldLineLevel = Line->Level;
3473 auto &Level = Line->Level;
3474
3475 if (IsGotoLabel && IndentGotoLabel == FormatStyle::IGLS_NoIndent)
3476 Level = 0;
3477
3478 if (!IsGotoLabel || IndentGotoLabel == FormatStyle::IGLS_OuterIndent) {
3479 if (OldLineLevel > 1 || (!Line->InPPDirective && OldLineLevel > 0))
3480 --Level;
3481 }
3482
3483 if (!IsGotoLabel && !Style.IndentCaseBlocks &&
3484 CommentsBeforeNextToken.empty() && FormatTok->is(tok::l_brace)) {
3485 CompoundStatementIndenter Indenter(this, Level,
3486 Style.BraceWrapping.AfterCaseLabel,
3487 Style.BraceWrapping.IndentBraces);
3488 parseBlock();
3489 if (FormatTok->is(tok::kw_break)) {
3490 if (Style.BraceWrapping.AfterControlStatement ==
3492 addUnwrappedLine();
3493 if (!Style.IndentCaseBlocks &&
3494 Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths) {
3495 ++Level;
3496 }
3497 }
3498 parseStructuralElement();
3499 }
3500 addUnwrappedLine();
3501 } else {
3502 if (FormatTok->is(tok::semi))
3503 nextToken();
3504 addUnwrappedLine();
3505 }
3506
3507 Level = OldLineLevel;
3508
3509 if (FormatTok->isNot(tok::l_brace)) {
3510 parseStructuralElement();
3511 addUnwrappedLine();
3512 }
3513}
3514
3515void UnwrappedLineParser::parseCaseLabel() {
3516 assert(FormatTok->is(tok::kw_case) && "'case' expected");
3517 auto *Case = FormatTok;
3518
3519 // FIXME: fix handling of complex expressions here.
3520 do {
3521 nextToken();
3522 if (FormatTok->is(tok::colon)) {
3523 FormatTok->setFinalizedType(TT_CaseLabelColon);
3524 break;
3525 }
3526 if (Style.isJava() && FormatTok->is(tok::arrow)) {
3527 FormatTok->setFinalizedType(TT_CaseLabelArrow);
3528 Case->setFinalizedType(TT_SwitchExpressionLabel);
3529 break;
3530 }
3531 } while (!eof());
3532 parseLabel();
3533}
3534
3535void UnwrappedLineParser::parseSwitch(bool IsExpr) {
3536 assert(FormatTok->is(tok::kw_switch) && "'switch' expected");
3537 nextToken();
3538 if (FormatTok->is(tok::l_paren))
3539 parseParens();
3540
3541 keepAncestorBraces();
3542
3543 if (FormatTok->is(tok::l_brace)) {
3544 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3545 FormatTok->setFinalizedType(IsExpr ? TT_SwitchExpressionLBrace
3546 : TT_ControlStatementLBrace);
3547 if (IsExpr)
3548 parseChildBlock();
3549 else
3550 parseBlock();
3551 setPreviousRBraceType(TT_ControlStatementRBrace);
3552 if (!IsExpr)
3553 addUnwrappedLine();
3554 } else {
3555 addUnwrappedLine();
3556 ++Line->Level;
3557 parseStructuralElement();
3558 --Line->Level;
3559 }
3560
3561 if (Style.RemoveBracesLLVM)
3562 NestedTooDeep.pop_back();
3563}
3564
3565void UnwrappedLineParser::parseAccessSpecifier() {
3566 nextToken();
3567 // Understand Qt's slots.
3568 if (FormatTok->isOneOf(Keywords.kw_slots, Keywords.kw_qslots))
3569 nextToken();
3570 // Otherwise, we don't know what it is, and we'd better keep the next token.
3571 if (FormatTok->is(tok::colon))
3572 nextToken();
3573 addUnwrappedLine();
3574}
3575
3576/// Parses a requires, decides if it is a clause or an expression.
3577/// \pre The current token has to be the requires keyword.
3578/// \returns true if it parsed a clause.
3579bool UnwrappedLineParser::parseRequires(bool SeenEqual) {
3580 assert(FormatTok->is(tok::kw_requires) && "'requires' expected");
3581
3582 // We try to guess if it is a requires clause, or a requires expression. For
3583 // that we first check the next token.
3584 switch (Tokens->peekNextToken(/*SkipComment=*/true)->Tok.getKind()) {
3585 case tok::l_brace:
3586 // This can only be an expression, never a clause.
3587 parseRequiresExpression();
3588 return false;
3589 case tok::l_paren:
3590 // Clauses and expression can start with a paren, it's unclear what we have.
3591 break;
3592 default:
3593 // All other tokens can only be a clause.
3594 parseRequiresClause();
3595 return true;
3596 }
3597
3598 // Looking forward we would have to decide if there are function declaration
3599 // like arguments to the requires expression:
3600 // requires (T t) {
3601 // Or there is a constraint expression for the requires clause:
3602 // requires (C<T> && ...
3603
3604 // But first let's look behind.
3605 auto *PreviousNonComment = FormatTok->getPreviousNonComment();
3606
3607 if (!PreviousNonComment ||
3608 PreviousNonComment->is(TT_RequiresExpressionLBrace)) {
3609 // If there is no token, or an expression left brace, we are a requires
3610 // clause within a requires expression.
3611 parseRequiresClause();
3612 return true;
3613 }
3614
3615 switch (PreviousNonComment->Tok.getKind()) {
3616 case tok::greater:
3617 case tok::r_paren:
3618 case tok::kw_noexcept:
3619 case tok::kw_const:
3620 case tok::star:
3621 case tok::amp:
3622 // This is a requires clause.
3623 parseRequiresClause();
3624 return true;
3625 case tok::ampamp: {
3626 // This can be either:
3627 // if (... && requires (T t) ...)
3628 // Or
3629 // void member(...) && requires (C<T> ...
3630 // We check the one token before that for a const:
3631 // void member(...) const && requires (C<T> ...
3632 auto PrevPrev = PreviousNonComment->getPreviousNonComment();
3633 if ((PrevPrev && PrevPrev->is(tok::kw_const)) || !SeenEqual) {
3634 parseRequiresClause();
3635 return true;
3636 }
3637 break;
3638 }
3639 default:
3640 if (PreviousNonComment->isTypeOrIdentifier(LangOpts)) {
3641 // This is a requires clause.
3642 parseRequiresClause();
3643 return true;
3644 }
3645 // It's an expression.
3646 parseRequiresExpression();
3647 return false;
3648 }
3649
3650 // Now we look forward and try to check if the paren content is a parameter
3651 // list. The parameters can be cv-qualified and contain references or
3652 // pointers.
3653 // So we want basically to check for TYPE NAME, but TYPE can contain all kinds
3654 // of stuff: typename, const, *, &, &&, ::, identifiers.
3655
3656 unsigned StoredPosition = Tokens->getPosition();
3657 FormatToken *NextToken = Tokens->getNextToken();
3658 int Lookahead = 0;
3659 auto PeekNext = [&Lookahead, &NextToken, this] {
3660 ++Lookahead;
3661 NextToken = Tokens->getNextToken();
3662 };
3663
3664 bool FoundType = false;
3665 bool LastWasColonColon = false;
3666 int OpenAngles = 0;
3667
3668 for (; Lookahead < 50; PeekNext()) {
3669 switch (NextToken->Tok.getKind()) {
3670 case tok::kw_volatile:
3671 case tok::kw_const:
3672 case tok::comma:
3673 if (OpenAngles == 0) {
3674 FormatTok = Tokens->setPosition(StoredPosition);
3675 parseRequiresExpression();
3676 return false;
3677 }
3678 break;
3679 case tok::eof:
3680 // Break out of the loop.
3681 Lookahead = 50;
3682 break;
3683 case tok::coloncolon:
3684 LastWasColonColon = true;
3685 break;
3686 case tok::kw_decltype:
3687 case tok::identifier:
3688 if (FoundType && !LastWasColonColon && OpenAngles == 0) {
3689 FormatTok = Tokens->setPosition(StoredPosition);
3690 parseRequiresExpression();
3691 return false;
3692 }
3693 FoundType = true;
3694 LastWasColonColon = false;
3695 break;
3696 case tok::less:
3697 ++OpenAngles;
3698 break;
3699 case tok::greater:
3700 --OpenAngles;
3701 break;
3702 default:
3703 if (NextToken->isTypeName(LangOpts)) {
3704 FormatTok = Tokens->setPosition(StoredPosition);
3705 parseRequiresExpression();
3706 return false;
3707 }
3708 break;
3709 }
3710 }
3711 // This seems to be a complicated expression, just assume it's a clause.
3712 FormatTok = Tokens->setPosition(StoredPosition);
3713 parseRequiresClause();
3714 return true;
3715}
3716
3717/// Parses a requires clause.
3718/// \sa parseRequiresExpression
3719///
3720/// Returns if it either has finished parsing the clause, or it detects, that
3721/// the clause is incorrect.
3722void UnwrappedLineParser::parseRequiresClause() {
3723 assert(FormatTok->is(tok::kw_requires) && "'requires' expected");
3724
3725 // If there is no previous token, we are within a requires expression,
3726 // otherwise we will always have the template or function declaration in front
3727 // of it.
3728 bool InRequiresExpression =
3729 !FormatTok->Previous ||
3730 FormatTok->Previous->is(TT_RequiresExpressionLBrace);
3731
3732 FormatTok->setFinalizedType(InRequiresExpression
3733 ? TT_RequiresClauseInARequiresExpression
3734 : TT_RequiresClause);
3735 nextToken();
3736
3737 // NOTE: parseConstraintExpression is only ever called from this function.
3738 // It could be inlined into here.
3739 parseConstraintExpression();
3740
3741 if (!InRequiresExpression && FormatTok->Previous)
3742 FormatTok->Previous->ClosesRequiresClause = true;
3743}
3744
3745/// Parses a requires expression.
3746/// \sa parseRequiresClause
3747///
3748/// Returns if it either has finished parsing the expression, or it detects,
3749/// that the expression is incorrect.
3750void UnwrappedLineParser::parseRequiresExpression() {
3751 assert(FormatTok->is(tok::kw_requires) && "'requires' expected");
3752
3753 FormatTok->setFinalizedType(TT_RequiresExpression);
3754 nextToken();
3755
3756 if (FormatTok->is(tok::l_paren)) {
3757 FormatTok->setFinalizedType(TT_RequiresExpressionLParen);
3758 parseParens();
3759 }
3760
3761 if (FormatTok->is(tok::l_brace)) {
3762 FormatTok->setFinalizedType(TT_RequiresExpressionLBrace);
3763 parseChildBlock();
3764 }
3765}
3766
3767/// Parses a constraint expression.
3768///
3769/// This is the body of a requires clause. It returns, when the parsing is
3770/// complete, or the expression is incorrect.
3771void UnwrappedLineParser::parseConstraintExpression() {
3772 // The special handling for lambdas is needed since tryToParseLambda() eats a
3773 // token and if a requires expression is the last part of a requires clause
3774 // and followed by an attribute like [[nodiscard]] the ClosesRequiresClause is
3775 // not set on the correct token. Thus we need to be aware if we even expect a
3776 // lambda to be possible.
3777 // template <typename T> requires requires { ... } [[nodiscard]] ...;
3778 bool LambdaNextTimeAllowed = true;
3779
3780 // Within lambda declarations, it is permitted to put a requires clause after
3781 // its template parameter list, which would place the requires clause right
3782 // before the parentheses of the parameters of the lambda declaration. Thus,
3783 // we track if we expect to see grouping parentheses at all.
3784 // Without this check, `requires foo<T> (T t)` in the below example would be
3785 // seen as the whole requires clause, accidentally eating the parameters of
3786 // the lambda.
3787 // [&]<typename T> requires foo<T> (T t) { ... };
3788 bool TopLevelParensAllowed = true;
3789
3790 do {
3791 bool LambdaThisTimeAllowed = std::exchange(LambdaNextTimeAllowed, false);
3792
3793 switch (FormatTok->Tok.getKind()) {
3794 case tok::kw_requires:
3795 parseRequiresExpression();
3796 break;
3797
3798 case tok::l_paren:
3799 if (!TopLevelParensAllowed)
3800 return;
3801 parseParens(/*AmpAmpTokenType=*/TT_BinaryOperator);
3802 TopLevelParensAllowed = false;
3803 break;
3804
3805 case tok::l_square:
3806 if (!LambdaThisTimeAllowed || !tryToParseLambda())
3807 return;
3808 break;
3809
3810 case tok::kw_const:
3811 case tok::semi:
3812 case tok::kw_class:
3813 case tok::kw_struct:
3814 case tok::kw_union:
3815 return;
3816
3817 case tok::l_brace:
3818 // Potential function body.
3819 return;
3820
3821 case tok::ampamp:
3822 case tok::pipepipe:
3823 FormatTok->setFinalizedType(TT_BinaryOperator);
3824 nextToken();
3825 LambdaNextTimeAllowed = true;
3826 TopLevelParensAllowed = true;
3827 break;
3828
3829 case tok::comma:
3830 case tok::comment:
3831 LambdaNextTimeAllowed = LambdaThisTimeAllowed;
3832 nextToken();
3833 break;
3834
3835 case tok::kw_sizeof:
3836 case tok::greater:
3837 case tok::greaterequal:
3838 case tok::greatergreater:
3839 case tok::less:
3840 case tok::lessequal:
3841 case tok::lessless:
3842 case tok::equalequal:
3843 case tok::exclaim:
3844 case tok::exclaimequal:
3845 case tok::plus:
3846 case tok::minus:
3847 case tok::star:
3848 case tok::slash:
3849 LambdaNextTimeAllowed = true;
3850 TopLevelParensAllowed = true;
3851 // Just eat them.
3852 nextToken();
3853 break;
3854
3855 case tok::numeric_constant:
3856 case tok::coloncolon:
3857 case tok::kw_true:
3858 case tok::kw_false:
3859 TopLevelParensAllowed = false;
3860 // Just eat them.
3861 nextToken();
3862 break;
3863
3864 case tok::kw_static_cast:
3865 case tok::kw_const_cast:
3866 case tok::kw_reinterpret_cast:
3867 case tok::kw_dynamic_cast:
3868 nextToken();
3869 if (FormatTok->isNot(tok::less))
3870 return;
3871
3872 nextToken();
3873 parseBracedList(/*IsAngleBracket=*/true);
3874 break;
3875
3876 default:
3877 if (!FormatTok->Tok.getIdentifierInfo()) {
3878 // Identifiers are part of the default case, we check for more then
3879 // tok::identifier to handle builtin type traits.
3880 return;
3881 }
3882
3883 // We need to differentiate identifiers for a template deduction guide,
3884 // variables, or function return types (the constraint expression has
3885 // ended before that), and basically all other cases. But it's easier to
3886 // check the other way around.
3887 assert(FormatTok->Previous);
3888 switch (FormatTok->Previous->Tok.getKind()) {
3889 case tok::coloncolon: // Nested identifier.
3890 case tok::ampamp: // Start of a function or variable for the
3891 case tok::pipepipe: // constraint expression. (binary)
3892 case tok::exclaim: // The same as above, but unary.
3893 case tok::kw_requires: // Initial identifier of a requires clause.
3894 case tok::equal: // Initial identifier of a concept declaration.
3895 case tok::kw_template: // A dependent template.
3896 break;
3897 default:
3898 return;
3899 }
3900
3901 // Read identifier with optional template declaration.
3902 nextToken();
3903 if (FormatTok->is(tok::less)) {
3904 nextToken();
3905 parseBracedList(/*IsAngleBracket=*/true);
3906 }
3907 TopLevelParensAllowed = false;
3908 break;
3909 }
3910 } while (!eof());
3911}
3912
3913bool UnwrappedLineParser::parseEnum() {
3914 const FormatToken &InitialToken = *FormatTok;
3915
3916 // Won't be 'enum' for NS_ENUMs.
3917 if (FormatTok->is(tok::kw_enum))
3918 nextToken();
3919
3920 // In TypeScript, "enum" can also be used as property name, e.g. in interface
3921 // declarations. An "enum" keyword followed by a colon would be a syntax
3922 // error and thus assume it is just an identifier.
3923 if (Style.isJavaScript() && FormatTok->isOneOf(tok::colon, tok::question))
3924 return false;
3925
3926 // In protobuf, "enum" can be used as a field name.
3927 if (Style.Language == FormatStyle::LK_Proto && FormatTok->is(tok::equal))
3928 return false;
3929
3930 if (IsCpp) {
3931 // Eat up enum class ...
3932 if (FormatTok->isOneOf(tok::kw_class, tok::kw_struct))
3933 nextToken();
3934 while (FormatTok->is(tok::l_square))
3935 if (!handleCppAttributes())
3936 return false;
3937 }
3938
3939 while (FormatTok->Tok.getIdentifierInfo() ||
3940 FormatTok->isOneOf(tok::colon, tok::coloncolon, tok::less,
3941 tok::greater, tok::comma, tok::question,
3942 tok::l_square)) {
3943 if (FormatTok->is(tok::colon))
3944 FormatTok->setFinalizedType(TT_EnumUnderlyingTypeColon);
3945 if (Style.isVerilog()) {
3946 FormatTok->setFinalizedType(TT_VerilogDimensionedTypeName);
3947 nextToken();
3948 // In Verilog the base type can have dimensions.
3949 while (FormatTok->is(tok::l_square))
3950 parseSquare();
3951 } else {
3952 nextToken();
3953 }
3954 // We can have macros or attributes in between 'enum' and the enum name.
3955 if (FormatTok->is(tok::l_paren))
3956 parseParens();
3957 if (FormatTok->is(tok::identifier)) {
3958 nextToken();
3959 // If there are two identifiers in a row, this is likely an elaborate
3960 // return type. In Java, this can be "implements", etc.
3961 if (IsCpp && FormatTok->is(tok::identifier))
3962 return false;
3963 }
3964 }
3965
3966 // Just a declaration or something is wrong.
3967 if (FormatTok->isNot(tok::l_brace))
3968 return true;
3969 FormatTok->setFinalizedType(TT_EnumLBrace);
3970 FormatTok->setBlockKind(BK_Block);
3971
3972 if (Style.isJava()) {
3973 // Java enums are different.
3974 parseJavaEnumBody();
3975 return true;
3976 }
3977 if (Style.Language == FormatStyle::LK_Proto) {
3978 parseBlock(/*MustBeDeclaration=*/true);
3979 return true;
3980 }
3981
3982 const bool ManageWhitesmithsBraces =
3983 Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths;
3984
3985 if (!Style.AllowShortEnumsOnASingleLine &&
3986 ShouldBreakBeforeBrace(Style, InitialToken,
3987 Tokens->peekNextToken()->is(tok::r_brace))) {
3988 addUnwrappedLine();
3989
3990 // If we're in Whitesmiths mode, indent the brace if we're not indenting
3991 // the whole block.
3992 if (ManageWhitesmithsBraces)
3993 ++Line->Level;
3994 }
3995 // Parse enum body.
3996 nextToken();
3997 if (!Style.AllowShortEnumsOnASingleLine) {
3998 addUnwrappedLine();
3999 if (!ManageWhitesmithsBraces)
4000 ++Line->Level;
4001 }
4002 const auto OpeningLineIndex = CurrentLines->empty()
4003 ? UnwrappedLine::kInvalidIndex
4004 : CurrentLines->size() - 1;
4005 bool HasError = !parseBracedList(/*IsAngleBracket=*/false, /*IsEnum=*/true);
4006 if (!Style.AllowShortEnumsOnASingleLine && !ManageWhitesmithsBraces)
4007 --Line->Level;
4008 if (HasError) {
4009 if (FormatTok->is(tok::semi))
4010 nextToken();
4011 addUnwrappedLine();
4012 }
4013 setPreviousRBraceType(TT_EnumRBrace);
4014 if (ManageWhitesmithsBraces)
4015 Line->MatchingOpeningBlockLineIndex = OpeningLineIndex;
4016 return true;
4017
4018 // There is no addUnwrappedLine() here so that we fall through to parsing a
4019 // structural element afterwards. Thus, in "enum A {} n, m;",
4020 // "} n, m;" will end up in one unwrapped line.
4021}
4022
4023bool UnwrappedLineParser::parseStructLike() {
4024 // parseRecord falls through and does not yet add an unwrapped line as a
4025 // record declaration or definition can start a structural element.
4026 parseRecord();
4027 // This does not apply to Java, JavaScript and C#.
4028 if (Style.isJava() || Style.isJavaScript() || Style.isCSharp()) {
4029 if (FormatTok->is(tok::semi))
4030 nextToken();
4031 addUnwrappedLine();
4032 return true;
4033 }
4034 return false;
4035}
4036
4037namespace {
4038// A class used to set and restore the Token position when peeking
4039// ahead in the token source.
4040class ScopedTokenPosition {
4041 unsigned StoredPosition;
4042 FormatTokenSource *Tokens;
4043
4044public:
4045 ScopedTokenPosition(FormatTokenSource *Tokens) : Tokens(Tokens) {
4046 assert(Tokens && "Tokens expected to not be null");
4047 StoredPosition = Tokens->getPosition();
4048 }
4049
4050 ~ScopedTokenPosition() { Tokens->setPosition(StoredPosition); }
4051};
4052} // namespace
4053
4054// Look to see if we have [[ by looking ahead, if
4055// its not then rewind to the original position.
4056bool UnwrappedLineParser::tryToParseSimpleAttribute() {
4057 ScopedTokenPosition AutoPosition(Tokens);
4058 FormatToken *Tok = Tokens->getNextToken();
4059 // We already read the first [ check for the second.
4060 if (Tok->isNot(tok::l_square))
4061 return false;
4062 // Double check that the attribute is just something
4063 // fairly simple.
4064 while (Tok->isNot(tok::eof)) {
4065 if (Tok->is(tok::r_square))
4066 break;
4067 Tok = Tokens->getNextToken();
4068 }
4069 if (Tok->is(tok::eof))
4070 return false;
4071 Tok = Tokens->getNextToken();
4072 if (Tok->isNot(tok::r_square))
4073 return false;
4074 Tok = Tokens->getNextToken();
4075 if (Tok->is(tok::semi))
4076 return false;
4077 return true;
4078}
4079
4080void UnwrappedLineParser::parseJavaEnumBody() {
4081 assert(FormatTok->is(tok::l_brace));
4082 const FormatToken *OpeningBrace = FormatTok;
4083
4084 // Determine whether the enum is simple, i.e. does not have a semicolon or
4085 // constants with class bodies. Simple enums can be formatted like braced
4086 // lists, contracted to a single line, etc.
4087 unsigned StoredPosition = Tokens->getPosition();
4088 bool IsSimple = true;
4089 FormatToken *Tok = Tokens->getNextToken();
4090 while (Tok->isNot(tok::eof)) {
4091 if (Tok->is(tok::r_brace))
4092 break;
4093 if (Tok->isOneOf(tok::l_brace, tok::semi)) {
4094 IsSimple = false;
4095 break;
4096 }
4097 // FIXME: This will also mark enums with braces in the arguments to enum
4098 // constants as "not simple". This is probably fine in practice, though.
4099 Tok = Tokens->getNextToken();
4100 }
4101 FormatTok = Tokens->setPosition(StoredPosition);
4102
4103 if (IsSimple) {
4104 nextToken();
4105 parseBracedList();
4106 addUnwrappedLine();
4107 return;
4108 }
4109
4110 // Parse the body of a more complex enum.
4111 // First add a line for everything up to the "{".
4112 nextToken();
4113 addUnwrappedLine();
4114 ++Line->Level;
4115
4116 // Parse the enum constants.
4117 while (!eof()) {
4118 if (FormatTok->is(tok::l_brace)) {
4119 // Parse the constant's class body.
4120 parseBlock(/*MustBeDeclaration=*/true, /*AddLevels=*/1u,
4121 /*MunchSemi=*/false);
4122 } else if (FormatTok->is(tok::l_paren)) {
4123 parseParens();
4124 } else if (FormatTok->is(tok::comma)) {
4125 nextToken();
4126 addUnwrappedLine();
4127 } else if (FormatTok->is(tok::semi)) {
4128 nextToken();
4129 addUnwrappedLine();
4130 break;
4131 } else if (FormatTok->is(tok::r_brace)) {
4132 addUnwrappedLine();
4133 break;
4134 } else {
4135 nextToken();
4136 }
4137 }
4138
4139 // Parse the class body after the enum's ";" if any.
4140 parseLevel(OpeningBrace);
4141 nextToken();
4142 --Line->Level;
4143 addUnwrappedLine();
4144}
4145
4146void UnwrappedLineParser::parseRecord(bool ParseAsExpr, bool IsJavaRecord) {
4147 assert(!IsJavaRecord || FormatTok->is(Keywords.kw_record));
4148 const FormatToken &InitialToken = *FormatTok;
4149 nextToken();
4150
4151 FormatToken *ClassName =
4152 IsJavaRecord && FormatTok->is(tok::identifier) ? FormatTok : nullptr;
4153 bool IsDerived = false;
4154 auto IsNonMacroIdentifier = [](const FormatToken *Tok) {
4155 return Tok->is(tok::identifier) && Tok->TokenText != Tok->TokenText.upper();
4156 };
4157 // JavaScript/TypeScript supports anonymous classes like:
4158 // a = class extends foo { }
4159 bool JSPastExtendsOrImplements = false;
4160 // The actual identifier can be a nested name specifier, and in macros
4161 // it is often token-pasted.
4162 // An [[attribute]] can be before the identifier.
4163 while (FormatTok->isOneOf(tok::identifier, tok::coloncolon, tok::hashhash,
4164 tok::kw_alignas, tok::l_square) ||
4165 FormatTok->isAttribute() ||
4166 ((Style.isJava() || Style.isJavaScript()) &&
4167 FormatTok->isOneOf(tok::period, tok::comma)) ||
4168 (Style.isVerilog() &&
4169 FormatTok->isOneOf(tok::kw_signed, tok::kw_unsigned))) {
4170 if (Style.isJavaScript() &&
4171 FormatTok->isOneOf(Keywords.kw_extends, Keywords.kw_implements)) {
4172 JSPastExtendsOrImplements = true;
4173 // JavaScript/TypeScript supports inline object types in
4174 // extends/implements positions:
4175 // class Foo implements {bar: number} { }
4176 nextToken();
4177 if (FormatTok->is(tok::l_brace)) {
4178 tryToParseBracedList();
4179 continue;
4180 }
4181 }
4182 if (FormatTok->is(tok::l_square) && handleCppAttributes())
4183 continue;
4184 auto *Previous = FormatTok;
4185 nextToken();
4186 switch (FormatTok->Tok.getKind()) {
4187 case tok::l_paren:
4188 // We can have macros in between 'class' and the class name.
4189 if (IsJavaRecord || !IsNonMacroIdentifier(Previous) ||
4190 // e.g. `struct macro(a) S { int i; };`
4191 Previous->Previous == &InitialToken) {
4192 parseParens();
4193 }
4194 break;
4195 case tok::coloncolon:
4196 case tok::hashhash:
4197 break;
4198 default:
4199 if (JSPastExtendsOrImplements || ClassName ||
4200 Previous->isNot(tok::identifier) || Previous->is(TT_AttributeMacro)) {
4201 break;
4202 }
4203 if (const auto Text = Previous->TokenText;
4204 Text.size() == 1 || Text != Text.upper()) {
4205 ClassName = Previous;
4206 }
4207 }
4208 }
4209
4210 auto IsListInitialization = [&] {
4211 if (!ClassName || IsDerived || JSPastExtendsOrImplements)
4212 return false;
4213 assert(FormatTok->is(tok::l_brace));
4214 const auto *Prev = FormatTok->getPreviousNonComment();
4215 assert(Prev);
4216 return Prev != ClassName && Prev->is(tok::identifier) &&
4217 Prev->isNot(Keywords.kw_final) && tryToParseBracedList();
4218 };
4219
4220 if (FormatTok->isOneOf(tok::colon, tok::less)) {
4221 int AngleNestingLevel = 0;
4222 do {
4223 if (FormatTok->is(tok::less))
4224 ++AngleNestingLevel;
4225 else if (FormatTok->is(tok::greater))
4226 --AngleNestingLevel;
4227
4228 if (AngleNestingLevel == 0) {
4229 if (FormatTok->is(tok::colon)) {
4230 IsDerived = true;
4231 } else if (!IsDerived && FormatTok->is(tok::identifier) &&
4232 FormatTok->Previous->is(tok::coloncolon)) {
4233 ClassName = FormatTok;
4234 } else if (FormatTok->is(tok::l_paren) &&
4235 IsNonMacroIdentifier(FormatTok->Previous)) {
4236 break;
4237 }
4238 }
4239 if (FormatTok->is(tok::l_brace)) {
4240 if (AngleNestingLevel == 0 && IsListInitialization())
4241 return;
4242 calculateBraceTypes(/*ExpectClassBody=*/true);
4243 if (!tryToParseBracedList())
4244 break;
4245 }
4246 if (FormatTok->is(tok::l_square)) {
4247 FormatToken *Previous = FormatTok->Previous;
4248 if (!Previous || (Previous->isNot(tok::r_paren) &&
4249 !Previous->isTypeOrIdentifier(LangOpts))) {
4250 // Don't try parsing a lambda if we had a closing parenthesis before,
4251 // it was probably a pointer to an array: int (*)[].
4252 if (!tryToParseLambda())
4253 continue;
4254 } else {
4255 parseSquare();
4256 continue;
4257 }
4258 }
4259 if (FormatTok->is(tok::semi))
4260 return;
4261 if (Style.isCSharp() && FormatTok->is(Keywords.kw_where)) {
4262 addUnwrappedLine();
4263 nextToken();
4264 parseCSharpGenericTypeConstraint();
4265 break;
4266 }
4267 nextToken();
4268 } while (!eof());
4269 }
4270
4271 auto GetBraceTypes =
4272 [](const FormatToken &RecordTok) -> std::pair<TokenType, TokenType> {
4273 switch (RecordTok.Tok.getKind()) {
4274 case tok::kw_class:
4275 return {TT_ClassLBrace, TT_ClassRBrace};
4276 case tok::kw_struct:
4277 return {TT_StructLBrace, TT_StructRBrace};
4278 case tok::kw_union:
4279 return {TT_UnionLBrace, TT_UnionRBrace};
4280 default:
4281 // Useful for e.g. interface.
4282 return {TT_RecordLBrace, TT_RecordRBrace};
4283 }
4284 };
4285 if (FormatTok->is(tok::l_brace)) {
4286 if (IsListInitialization())
4287 return;
4288 if (ClassName)
4289 ClassName->setFinalizedType(TT_ClassHeadName);
4290 auto [OpenBraceType, ClosingBraceType] = GetBraceTypes(InitialToken);
4291 FormatTok->setFinalizedType(OpenBraceType);
4292 if (ParseAsExpr) {
4293 parseChildBlock();
4294 } else {
4295 if (ShouldBreakBeforeBrace(Style, InitialToken,
4296 Tokens->peekNextToken()->is(tok::r_brace),
4297 IsJavaRecord)) {
4298 addUnwrappedLine();
4299 }
4300
4301 unsigned AddLevels = Style.IndentAccessModifiers ? 2u : 1u;
4302 parseBlock(/*MustBeDeclaration=*/true, AddLevels, /*MunchSemi=*/false);
4303 }
4304 setPreviousRBraceType(ClosingBraceType);
4305 }
4306 // There is no addUnwrappedLine() here so that we fall through to parsing a
4307 // structural element afterwards. Thus, in "class A {} n, m;",
4308 // "} n, m;" will end up in one unwrapped line.
4309}
4310
4311void UnwrappedLineParser::parseObjCMethod() {
4312 assert(FormatTok->isOneOf(tok::l_paren, tok::identifier) &&
4313 "'(' or identifier expected.");
4314 do {
4315 if (FormatTok->is(tok::semi)) {
4316 nextToken();
4317 addUnwrappedLine();
4318 return;
4319 } else if (FormatTok->is(tok::l_brace)) {
4320 if (Style.BraceWrapping.AfterFunction)
4321 addUnwrappedLine();
4322 parseBlock();
4323 addUnwrappedLine();
4324 return;
4325 } else {
4326 nextToken();
4327 }
4328 } while (!eof());
4329}
4330
4331void UnwrappedLineParser::parseObjCProtocolList() {
4332 assert(FormatTok->is(tok::less) && "'<' expected.");
4333 do {
4334 nextToken();
4335 // Early exit in case someone forgot a close angle.
4336 if (FormatTok->isOneOf(tok::semi, tok::l_brace, tok::objc_end))
4337 return;
4338 } while (!eof() && FormatTok->isNot(tok::greater));
4339 nextToken(); // Skip '>'.
4340}
4341
4342void UnwrappedLineParser::parseObjCUntilAtEnd() {
4343 do {
4344 if (FormatTok->is(tok::objc_end)) {
4345 nextToken();
4346 addUnwrappedLine();
4347 break;
4348 }
4349 if (FormatTok->is(tok::l_brace)) {
4350 parseBlock();
4351 // In ObjC interfaces, nothing should be following the "}".
4352 addUnwrappedLine();
4353 } else if (FormatTok->is(tok::r_brace)) {
4354 // Ignore stray "}". parseStructuralElement doesn't consume them.
4355 nextToken();
4356 addUnwrappedLine();
4357 } else if (FormatTok->isOneOf(tok::minus, tok::plus)) {
4358 nextToken();
4359 if (FormatTok->isOneOf(tok::l_paren, tok::identifier))
4360 parseObjCMethod();
4361 } else {
4362 parseStructuralElement();
4363 }
4364 } while (!eof());
4365}
4366
4367void UnwrappedLineParser::parseObjCInterfaceOrImplementation() {
4368 assert(FormatTok->isOneOf(tok::objc_interface, tok::objc_implementation));
4369 nextToken();
4370 nextToken(); // interface name
4371
4372 // @interface can be followed by a lightweight generic
4373 // specialization list, then either a base class or a category.
4374 if (FormatTok->is(tok::less))
4375 parseObjCLightweightGenerics();
4376 if (FormatTok->is(tok::colon)) {
4377 nextToken();
4378 nextToken(); // base class name
4379 // The base class can also have lightweight generics applied to it.
4380 if (FormatTok->is(tok::less))
4381 parseObjCLightweightGenerics();
4382 } else if (FormatTok->is(tok::l_paren)) {
4383 // Skip category, if present.
4384 parseParens();
4385 }
4386
4387 if (FormatTok->is(tok::less))
4388 parseObjCProtocolList();
4389
4390 if (FormatTok->is(tok::l_brace)) {
4391 if (Style.BraceWrapping.AfterObjCDeclaration)
4392 addUnwrappedLine();
4393 parseBlock(/*MustBeDeclaration=*/true);
4394 }
4395
4396 // With instance variables, this puts '}' on its own line. Without instance
4397 // variables, this ends the @interface line.
4398 addUnwrappedLine();
4399
4400 parseObjCUntilAtEnd();
4401}
4402
4403void UnwrappedLineParser::parseObjCLightweightGenerics() {
4404 assert(FormatTok->is(tok::less));
4405 // Unlike protocol lists, generic parameterizations support
4406 // nested angles:
4407 //
4408 // @interface Foo<ValueType : id <NSCopying, NSSecureCoding>> :
4409 // NSObject <NSCopying, NSSecureCoding>
4410 //
4411 // so we need to count how many open angles we have left.
4412 unsigned NumOpenAngles = 1;
4413 do {
4414 nextToken();
4415 // Early exit in case someone forgot a close angle.
4416 if (FormatTok->isOneOf(tok::semi, tok::l_brace, tok::objc_end))
4417 break;
4418 if (FormatTok->is(tok::less)) {
4419 ++NumOpenAngles;
4420 } else if (FormatTok->is(tok::greater)) {
4421 assert(NumOpenAngles > 0 && "'>' makes NumOpenAngles negative");
4422 --NumOpenAngles;
4423 }
4424 } while (!eof() && NumOpenAngles != 0);
4425 nextToken(); // Skip '>'.
4426}
4427
4428// Returns true for the declaration/definition form of @protocol,
4429// false for the expression form.
4430bool UnwrappedLineParser::parseObjCProtocol() {
4431 assert(FormatTok->is(tok::objc_protocol));
4432 nextToken();
4433
4434 if (FormatTok->is(tok::l_paren)) {
4435 // The expression form of @protocol, e.g. "Protocol* p = @protocol(foo);".
4436 return false;
4437 }
4438
4439 // The definition/declaration form,
4440 // @protocol Foo
4441 // - (int)someMethod;
4442 // @end
4443
4444 nextToken(); // protocol name
4445
4446 if (FormatTok->is(tok::less))
4447 parseObjCProtocolList();
4448
4449 // Check for protocol declaration.
4450 if (FormatTok->is(tok::semi)) {
4451 nextToken();
4452 addUnwrappedLine();
4453 return true;
4454 }
4455
4456 addUnwrappedLine();
4457 parseObjCUntilAtEnd();
4458 return true;
4459}
4460
4461void UnwrappedLineParser::parseJavaScriptEs6ImportExport() {
4462 bool IsImport = FormatTok->is(Keywords.kw_import);
4463 assert(IsImport || FormatTok->is(tok::kw_export));
4464 nextToken();
4465
4466 // Consume the "default" in "export default class/function".
4467 if (FormatTok->is(tok::kw_default))
4468 nextToken();
4469
4470 // Consume "async function", "function" and "default function", so that these
4471 // get parsed as free-standing JS functions, i.e. do not require a trailing
4472 // semicolon.
4473 if (FormatTok->is(Keywords.kw_async))
4474 nextToken();
4475 if (FormatTok->is(Keywords.kw_function)) {
4476 nextToken();
4477 return;
4478 }
4479
4480 // For imports, `export *`, `export {...}`, consume the rest of the line up
4481 // to the terminating `;`. For everything else, just return and continue
4482 // parsing the structural element, i.e. the declaration or expression for
4483 // `export default`.
4484 if (!IsImport && FormatTok->isNoneOf(tok::l_brace, tok::star) &&
4485 !FormatTok->isStringLiteral() &&
4486 !(FormatTok->is(Keywords.kw_type) &&
4487 Tokens->peekNextToken()->isOneOf(tok::l_brace, tok::star))) {
4488 return;
4489 }
4490
4491 while (!eof()) {
4492 if (FormatTok->is(tok::semi))
4493 return;
4494 if (Line->Tokens.empty()) {
4495 // Common issue: Automatic Semicolon Insertion wrapped the line, so the
4496 // import statement should terminate.
4497 return;
4498 }
4499 if (FormatTok->is(tok::l_brace)) {
4500 FormatTok->setBlockKind(BK_Block);
4501 nextToken();
4502 parseBracedList();
4503 } else {
4504 nextToken();
4505 }
4506 }
4507}
4508
4509void UnwrappedLineParser::parseStatementMacro() {
4510 nextToken();
4511 if (FormatTok->is(tok::l_paren))
4512 parseParens();
4513 if (FormatTok->is(tok::semi))
4514 nextToken();
4515 addUnwrappedLine();
4516}
4517
4518void UnwrappedLineParser::parseVerilogHierarchyIdentifier() {
4519 // consume things like a::`b.c[d:e] or a::*
4520 while (true) {
4521 if (FormatTok->isOneOf(tok::star, tok::period, tok::periodstar,
4522 tok::coloncolon, tok::hash) ||
4523 Keywords.isVerilogIdentifier(*FormatTok)) {
4524 nextToken();
4525 } else if (FormatTok->is(tok::l_square)) {
4526 parseSquare();
4527 } else {
4528 break;
4529 }
4530 }
4531}
4532
4533void UnwrappedLineParser::parseVerilogSensitivityList() {
4534 if (FormatTok->isNot(tok::at))
4535 return;
4536 nextToken();
4537 // A block event expression has 2 at signs.
4538 if (FormatTok->is(tok::at))
4539 nextToken();
4540 switch (FormatTok->Tok.getKind()) {
4541 case tok::star:
4542 nextToken();
4543 break;
4544 case tok::l_paren:
4545 parseParens();
4546 break;
4547 default:
4548 parseVerilogHierarchyIdentifier();
4549 break;
4550 }
4551}
4552
4553unsigned UnwrappedLineParser::parseVerilogHierarchyHeader() {
4554 unsigned AddLevels = 0;
4555
4556 if (FormatTok->is(Keywords.kw_clocking)) {
4557 nextToken();
4558 if (Keywords.isVerilogIdentifier(*FormatTok))
4559 nextToken();
4560 parseVerilogSensitivityList();
4561 if (FormatTok->is(tok::semi))
4562 nextToken();
4563 } else if (FormatTok->isOneOf(tok::kw_case, Keywords.kw_casex,
4564 Keywords.kw_casez, Keywords.kw_randcase,
4565 Keywords.kw_randsequence)) {
4566 if (Style.IndentCaseLabels)
4567 AddLevels++;
4568 nextToken();
4569 if (FormatTok->is(tok::l_paren)) {
4570 FormatTok->setFinalizedType(TT_ConditionLParen);
4571 parseParens();
4572 }
4573 if (FormatTok->isOneOf(Keywords.kw_inside, Keywords.kw_matches))
4574 nextToken();
4575 // The case header has no semicolon.
4576 } else {
4577 // "module" etc.
4578 nextToken();
4579 // all the words like the name of the module and specifiers like
4580 // "automatic" and the width of function return type
4581 while (true) {
4582 if (FormatTok->is(tok::l_square)) {
4583 auto Prev = FormatTok->getPreviousNonComment();
4584 if (Prev && Keywords.isVerilogIdentifier(*Prev))
4585 Prev->setFinalizedType(TT_VerilogDimensionedTypeName);
4586 parseSquare();
4587 } else if (Keywords.isVerilogIdentifier(*FormatTok) ||
4588 FormatTok->isOneOf(tok::hash, tok::hashhash, tok::coloncolon,
4589 Keywords.kw_automatic, tok::kw_static)) {
4590 nextToken();
4591 } else {
4592 break;
4593 }
4594 }
4595
4596 auto NewLine = [this]() {
4597 addUnwrappedLine();
4598 Line->IsContinuation = true;
4599 };
4600
4601 // package imports
4602 while (FormatTok->is(Keywords.kw_import)) {
4603 NewLine();
4604 nextToken();
4605 parseVerilogHierarchyIdentifier();
4606 if (FormatTok->is(tok::semi))
4607 nextToken();
4608 }
4609
4610 // parameters and ports
4611 if (FormatTok->is(Keywords.kw_verilogHash)) {
4612 NewLine();
4613 nextToken();
4614 if (FormatTok->is(tok::l_paren)) {
4615 FormatTok->setFinalizedType(TT_VerilogMultiLineListLParen);
4616 parseParens();
4617 }
4618 }
4619 if (FormatTok->is(tok::l_paren)) {
4620 NewLine();
4621 FormatTok->setFinalizedType(TT_VerilogMultiLineListLParen);
4622 parseParens();
4623 }
4624
4625 // extends and implements
4626 if (FormatTok->is(Keywords.kw_extends)) {
4627 NewLine();
4628 nextToken();
4629 parseVerilogHierarchyIdentifier();
4630 if (FormatTok->is(tok::l_paren))
4631 parseParens();
4632 }
4633 if (FormatTok->is(Keywords.kw_implements)) {
4634 NewLine();
4635 do {
4636 nextToken();
4637 parseVerilogHierarchyIdentifier();
4638 } while (FormatTok->is(tok::comma));
4639 }
4640
4641 // Coverage event for cover groups.
4642 if (FormatTok->is(tok::at)) {
4643 NewLine();
4644 parseVerilogSensitivityList();
4645 }
4646
4647 if (FormatTok->is(tok::semi))
4648 nextToken(/*LevelDifference=*/1);
4649 addUnwrappedLine();
4650 }
4651
4652 return AddLevels;
4653}
4654
4655void UnwrappedLineParser::parseVerilogTable() {
4656 assert(FormatTok->is(Keywords.kw_table));
4657 nextToken(/*LevelDifference=*/1);
4658 addUnwrappedLine();
4659
4660 auto InitialLevel = Line->Level++;
4661 while (!eof() && !Keywords.isVerilogEnd(*FormatTok)) {
4662 FormatToken *Tok = FormatTok;
4663 nextToken();
4664 if (Tok->is(tok::semi))
4665 addUnwrappedLine();
4666 else if (Tok->isOneOf(tok::star, tok::colon, tok::question, tok::minus))
4667 Tok->setFinalizedType(TT_VerilogTableItem);
4668 }
4669 Line->Level = InitialLevel;
4670 nextToken(/*LevelDifference=*/-1);
4671 addUnwrappedLine();
4672}
4673
4674void UnwrappedLineParser::parseVerilogCaseLabel() {
4675 // The label will get unindented in AnnotatingParser. If there are no leading
4676 // spaces, indent the rest here so that things inside the block will be
4677 // indented relative to things outside. We don't use parseLabel because we
4678 // don't know whether this colon is a label or a ternary expression at this
4679 // point.
4680 auto OrigLevel = Line->Level;
4681 auto FirstLine = CurrentLines->size();
4682 if (Line->Level == 0 || (Line->InPPDirective && Line->Level <= 1))
4683 ++Line->Level;
4684 else if (!Style.IndentCaseBlocks && Keywords.isVerilogBegin(*FormatTok))
4685 --Line->Level;
4686 parseStructuralElement();
4687 // Restore the indentation in both the new line and the line that has the
4688 // label.
4689 if (CurrentLines->size() > FirstLine)
4690 (*CurrentLines)[FirstLine].Level = OrigLevel;
4691 Line->Level = OrigLevel;
4692}
4693
4694void UnwrappedLineParser::parseVerilogExtern() {
4695 assert(
4696 FormatTok->isOneOf(tok::kw_extern, tok::kw_export, Keywords.kw_import));
4697 nextToken();
4698 // "DPI-C"
4699 if (FormatTok->is(tok::string_literal))
4700 nextToken();
4701 skipVerilogQualifiers();
4702 if (Keywords.isVerilogIdentifier(*FormatTok))
4703 nextToken();
4704 if (FormatTok->is(tok::equal))
4705 nextToken();
4706 if (Keywords.isVerilogHierarchy(*FormatTok))
4707 parseVerilogHierarchyHeader();
4708}
4709
4710void UnwrappedLineParser::skipVerilogQualifiers() {
4711 while (FormatTok->isOneOf(tok::kw_protected, tok::kw_virtual, tok::kw_static,
4712 Keywords.kw_rand, Keywords.kw_context,
4713 Keywords.kw_pure, Keywords.kw_randc,
4714 Keywords.kw_local)) {
4715 nextToken();
4716 }
4717}
4718
4719bool UnwrappedLineParser::containsExpansion(const UnwrappedLine &Line) const {
4720 for (const auto &N : Line.Tokens) {
4721 if (N.Tok->MacroCtx)
4722 return true;
4723 for (const UnwrappedLine &Child : N.Children)
4724 if (containsExpansion(Child))
4725 return true;
4726 }
4727 return false;
4728}
4729
4730void UnwrappedLineParser::addUnwrappedLine(LineLevel AdjustLevel) {
4731 if (Line->Tokens.empty())
4732 return;
4733 LLVM_DEBUG({
4734 if (!parsingPPDirective()) {
4735 llvm::dbgs() << "Adding unwrapped line:\n";
4736 printDebugInfo(*Line);
4737 }
4738 });
4739
4740 // If this line closes a block when in Whitesmiths mode, remember that
4741 // information so that the level can be decreased after the line is added.
4742 // This has to happen after the addition of the line since the line itself
4743 // needs to be indented.
4744 bool ClosesWhitesmithsBlock =
4745 Line->MatchingOpeningBlockLineIndex != UnwrappedLine::kInvalidIndex &&
4746 Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths;
4747
4748 // If the current line was expanded from a macro call, we use it to
4749 // reconstruct an unwrapped line from the structure of the expanded unwrapped
4750 // line and the unexpanded token stream.
4751 if (!parsingPPDirective() && !InExpansion && containsExpansion(*Line)) {
4752 if (!Reconstruct)
4753 Reconstruct.emplace(Line->Level, Unexpanded);
4754 Reconstruct->addLine(*Line);
4755
4756 // While the reconstructed unexpanded lines are stored in the normal
4757 // flow of lines, the expanded lines are stored on the side to be analyzed
4758 // in an extra step.
4759 CurrentExpandedLines.push_back(std::move(*Line));
4760
4761 if (Reconstruct->finished()) {
4762 UnwrappedLine Reconstructed = std::move(*Reconstruct).takeResult();
4763 assert(!Reconstructed.Tokens.empty() &&
4764 "Reconstructed must at least contain the macro identifier.");
4765 assert(!parsingPPDirective());
4766 LLVM_DEBUG({
4767 llvm::dbgs() << "Adding unexpanded line:\n";
4768 printDebugInfo(Reconstructed);
4769 });
4770 ExpandedLines[Reconstructed.Tokens.begin()->Tok] = CurrentExpandedLines;
4771 Lines.push_back(std::move(Reconstructed));
4772 CurrentExpandedLines.clear();
4773 Reconstruct.reset();
4774 }
4775 } else {
4776 // At the top level we only get here when no unexpansion is going on, or
4777 // when conditional formatting led to unfinished macro reconstructions.
4778 assert(!Reconstruct || (CurrentLines != &Lines) || !PP.Stack.empty());
4779 CurrentLines->push_back(std::move(*Line));
4780 }
4781 Line->Tokens.clear();
4782 Line->MatchingOpeningBlockLineIndex = UnwrappedLine::kInvalidIndex;
4783 Line->FirstStartColumn = 0;
4784 Line->IsContinuation = false;
4785 Line->SeenDecltypeAuto = false;
4786 Line->IsModuleOrImportDecl = false;
4787
4788 if (ClosesWhitesmithsBlock && AdjustLevel == LineLevel::Remove)
4789 --Line->Level;
4790 if (!parsingPPDirective() && !PreprocessorDirectives.empty()) {
4791 CurrentLines->append(
4792 std::make_move_iterator(PreprocessorDirectives.begin()),
4793 std::make_move_iterator(PreprocessorDirectives.end()));
4794 PreprocessorDirectives.clear();
4795 }
4796 // Disconnect the current token from the last token on the previous line.
4797 FormatTok->Previous = nullptr;
4798}
4799
4800bool UnwrappedLineParser::eof() const { return FormatTok->is(tok::eof); }
4801
4802bool UnwrappedLineParser::isOnNewLine(const FormatToken &FormatTok) {
4803 return (Line->InPPDirective || FormatTok.HasUnescapedNewline) &&
4804 FormatTok.NewlinesBefore > 0;
4805}
4806
4807// Checks if \p FormatTok is a line comment that continues the line comment
4808// section on \p Line.
4809static bool
4811 const UnwrappedLine &Line, const FormatStyle &Style,
4812 const llvm::Regex &CommentPragmasRegex) {
4813 if (Line.Tokens.empty() || Style.ReflowComments != FormatStyle::RCS_Always)
4814 return false;
4815
4816 StringRef IndentContent = FormatTok.TokenText;
4817 if (FormatTok.TokenText.starts_with("//") ||
4818 FormatTok.TokenText.starts_with("/*")) {
4819 IndentContent = FormatTok.TokenText.substr(2);
4820 }
4821 if (CommentPragmasRegex.match(IndentContent))
4822 return false;
4823
4824 // If Line starts with a line comment, then FormatTok continues the comment
4825 // section if its original column is greater or equal to the original start
4826 // column of the line.
4827 //
4828 // Define the min column token of a line as follows: if a line ends in '{' or
4829 // contains a '{' followed by a line comment, then the min column token is
4830 // that '{'. Otherwise, the min column token of the line is the first token of
4831 // the line.
4832 //
4833 // If Line starts with a token other than a line comment, then FormatTok
4834 // continues the comment section if its original column is greater than the
4835 // original start column of the min column token of the line.
4836 //
4837 // For example, the second line comment continues the first in these cases:
4838 //
4839 // // first line
4840 // // second line
4841 //
4842 // and:
4843 //
4844 // // first line
4845 // // second line
4846 //
4847 // and:
4848 //
4849 // int i; // first line
4850 // // second line
4851 //
4852 // and:
4853 //
4854 // do { // first line
4855 // // second line
4856 // int i;
4857 // } while (true);
4858 //
4859 // and:
4860 //
4861 // enum {
4862 // a, // first line
4863 // // second line
4864 // b
4865 // };
4866 //
4867 // The second line comment doesn't continue the first in these cases:
4868 //
4869 // // first line
4870 // // second line
4871 //
4872 // and:
4873 //
4874 // int i; // first line
4875 // // second line
4876 //
4877 // and:
4878 //
4879 // do { // first line
4880 // // second line
4881 // int i;
4882 // } while (true);
4883 //
4884 // and:
4885 //
4886 // enum {
4887 // a, // first line
4888 // // second line
4889 // };
4890 const FormatToken *MinColumnToken = Line.Tokens.front().Tok;
4891
4892 // Scan for '{//'. If found, use the column of '{' as a min column for line
4893 // comment section continuation.
4894 const FormatToken *PreviousToken = nullptr;
4895 for (const UnwrappedLineNode &Node : Line.Tokens) {
4896 if (PreviousToken && PreviousToken->is(tok::l_brace) &&
4897 isLineComment(*Node.Tok)) {
4898 MinColumnToken = PreviousToken;
4899 break;
4900 }
4901 PreviousToken = Node.Tok;
4902
4903 // Grab the last newline preceding a token in this unwrapped line.
4904 if (Node.Tok->NewlinesBefore > 0)
4905 MinColumnToken = Node.Tok;
4906 }
4907 if (PreviousToken && PreviousToken->is(tok::l_brace))
4908 MinColumnToken = PreviousToken;
4909
4910 return continuesLineComment(FormatTok, /*Previous=*/Line.Tokens.back().Tok,
4911 MinColumnToken);
4912}
4913
4914void UnwrappedLineParser::flushComments(bool NewlineBeforeNext) {
4915 bool JustComments = Line->Tokens.empty();
4916 for (FormatToken *Tok : CommentsBeforeNextToken) {
4917 // Line comments that belong to the same line comment section are put on the
4918 // same line since later we might want to reflow content between them.
4919 // Additional fine-grained breaking of line comment sections is controlled
4920 // by the class BreakableLineCommentSection in case it is desirable to keep
4921 // several line comment sections in the same unwrapped line.
4922 //
4923 // FIXME: Consider putting separate line comment sections as children to the
4924 // unwrapped line instead.
4925 Tok->ContinuesLineCommentSection =
4926 continuesLineCommentSection(*Tok, *Line, Style, CommentPragmasRegex);
4927 if (isOnNewLine(*Tok) && JustComments && !Tok->ContinuesLineCommentSection)
4928 addUnwrappedLine();
4929 pushToken(Tok);
4930 }
4931 if (NewlineBeforeNext && JustComments)
4932 addUnwrappedLine();
4933 CommentsBeforeNextToken.clear();
4934}
4935
4936void UnwrappedLineParser::nextToken(int LevelDifference) {
4937 if (eof())
4938 return;
4939 flushComments(isOnNewLine(*FormatTok));
4940 pushToken(FormatTok);
4941 FormatToken *Previous = FormatTok;
4942 if (!Style.isJavaScript())
4943 readToken(LevelDifference);
4944 else
4945 readTokenWithJavaScriptASI();
4946 FormatTok->Previous = Previous;
4947 if (Style.isVerilog()) {
4948 // Blocks in Verilog can have `begin` and `end` instead of braces. For
4949 // keywords like `begin`, we can't treat them the same as left braces
4950 // because some contexts require one of them. For example structs use
4951 // braces and if blocks use keywords, and a left brace can occur in an if
4952 // statement, but it is not a block. For keywords like `end`, we simply
4953 // treat them the same as right braces.
4954 if (Keywords.isVerilogEnd(*FormatTok))
4955 FormatTok->Tok.setKind(tok::r_brace);
4956 }
4957}
4958
4959void UnwrappedLineParser::distributeComments(
4960 const ArrayRef<FormatToken *> &Comments, const FormatToken *NextTok) {
4961 // Whether or not a line comment token continues a line is controlled by
4962 // the method continuesLineCommentSection, with the following caveat:
4963 //
4964 // Define a trail of Comments to be a nonempty proper postfix of Comments such
4965 // that each comment line from the trail is aligned with the next token, if
4966 // the next token exists. If a trail exists, the beginning of the maximal
4967 // trail is marked as a start of a new comment section.
4968 //
4969 // For example in this code:
4970 //
4971 // int a; // line about a
4972 // // line 1 about b
4973 // // line 2 about b
4974 // int b;
4975 //
4976 // the two lines about b form a maximal trail, so there are two sections, the
4977 // first one consisting of the single comment "// line about a" and the
4978 // second one consisting of the next two comments.
4979 if (Comments.empty())
4980 return;
4981 bool ShouldPushCommentsInCurrentLine = true;
4982 bool HasTrailAlignedWithNextToken = false;
4983 unsigned StartOfTrailAlignedWithNextToken = 0;
4984 if (NextTok) {
4985 // We are skipping the first element intentionally.
4986 for (unsigned i = Comments.size() - 1; i > 0; --i) {
4987 if (Comments[i]->OriginalColumn == NextTok->OriginalColumn) {
4988 HasTrailAlignedWithNextToken = true;
4989 StartOfTrailAlignedWithNextToken = i;
4990 }
4991 }
4992 }
4993 for (unsigned i = 0, e = Comments.size(); i < e; ++i) {
4994 FormatToken *FormatTok = Comments[i];
4995 if (HasTrailAlignedWithNextToken && i == StartOfTrailAlignedWithNextToken) {
4996 FormatTok->ContinuesLineCommentSection = false;
4997 } else {
4998 FormatTok->ContinuesLineCommentSection = continuesLineCommentSection(
4999 *FormatTok, *Line, Style, CommentPragmasRegex);
5000 }
5001 if (!FormatTok->ContinuesLineCommentSection &&
5002 (isOnNewLine(*FormatTok) || FormatTok->IsFirst)) {
5003 ShouldPushCommentsInCurrentLine = false;
5004 }
5005 if (ShouldPushCommentsInCurrentLine)
5006 pushToken(FormatTok);
5007 else
5008 CommentsBeforeNextToken.push_back(FormatTok);
5009 }
5010}
5011
5012void UnwrappedLineParser::readToken(int LevelDifference) {
5014 bool PreviousWasComment = false;
5015 bool FirstNonCommentOnLine = false;
5016 do {
5017 FormatTok = Tokens->getNextToken();
5018 assert(FormatTok);
5019 while (FormatTok->isOneOf(TT_ConflictStart, TT_ConflictEnd,
5020 TT_ConflictAlternative)) {
5021 if (FormatTok->is(TT_ConflictStart))
5022 conditionalCompilationStart(/*Unreachable=*/false);
5023 else if (FormatTok->is(TT_ConflictAlternative))
5024 conditionalCompilationAlternative();
5025 else if (FormatTok->is(TT_ConflictEnd))
5026 conditionalCompilationEnd();
5027 FormatTok = Tokens->getNextToken();
5028 FormatTok->MustBreakBefore = true;
5029 FormatTok->MustBreakBeforeFinalized = true;
5030 }
5031
5032 auto IsFirstNonCommentOnLine = [](bool FirstNonCommentOnLine,
5033 const FormatToken &Tok,
5034 bool PreviousWasComment) {
5035 auto IsFirstOnLine = [](const FormatToken &Tok) {
5036 return Tok.HasUnescapedNewline || Tok.IsFirst;
5037 };
5038
5039 // Consider preprocessor directives preceded by block comments as first
5040 // on line.
5041 if (PreviousWasComment)
5042 return FirstNonCommentOnLine || IsFirstOnLine(Tok);
5043 return IsFirstOnLine(Tok);
5044 };
5045
5046 FirstNonCommentOnLine = IsFirstNonCommentOnLine(
5047 FirstNonCommentOnLine, *FormatTok, PreviousWasComment);
5048 PreviousWasComment = FormatTok->is(tok::comment);
5049
5050 while (!Line->InPPDirective && FormatTok->is(tok::hash) &&
5051 FirstNonCommentOnLine) {
5052 // In Verilog, the backtick is used for macro invocations. In TableGen,
5053 // the single hash is used for the paste operator.
5054 const auto *Next = Tokens->peekNextToken();
5055 if ((Style.isVerilog() && !Keywords.isVerilogPPDirective(*Next)) ||
5056 (Style.isTableGen() &&
5057 Next->isNoneOf(tok::kw_else, tok::pp_define, tok::pp_ifdef,
5058 tok::pp_ifndef, tok::pp_endif))) {
5059 break;
5060 }
5061 distributeComments(Comments, FormatTok);
5062 Comments.clear();
5063 // If the directive was parsed before the token stream was rewound (see
5064 // parseMacroCall()), its lines were kept. Parse it again only for its
5065 // effect on the preprocessor bookkeeping and discard the new lines.
5066 const bool ParsedBefore = !ParsedPPDirectives.insert(FormatTok).second;
5067 // If there is an unfinished unwrapped line, we flush the preprocessor
5068 // directives only after that unwrapped line was finished later.
5069 bool SwitchToPreprocessorLines = !Line->Tokens.empty();
5070 ScopedLineState BlockState(*this, SwitchToPreprocessorLines,
5071 /*DiscardLines=*/ParsedBefore);
5072 assert((LevelDifference >= 0 ||
5073 static_cast<unsigned>(-LevelDifference) <= Line->Level) &&
5074 "LevelDifference makes Line->Level negative");
5075 Line->Level += LevelDifference;
5076 // Comments stored before the preprocessor directive need to be output
5077 // before the preprocessor directive, at the same level as the
5078 // preprocessor directive, as we consider them to apply to the directive.
5079 if (Style.IndentPPDirectives == FormatStyle::PPDIS_BeforeHash &&
5080 PP.BranchLevel > 0) {
5081 Line->Level += PP.BranchLevel;
5082 }
5083 assert(Line->Level >= Line->UnbracedBodyLevel);
5084 Line->Level -= Line->UnbracedBodyLevel;
5085 flushComments(isOnNewLine(*FormatTok));
5086 const bool IsEndIf = Tokens->peekNextToken()->is(tok::pp_endif);
5087 parsePPDirective();
5088 PreviousWasComment = FormatTok->is(tok::comment);
5089 FirstNonCommentOnLine = IsFirstNonCommentOnLine(
5090 FirstNonCommentOnLine, *FormatTok, PreviousWasComment);
5091 // If the #endif of a potential include guard is the last thing in the
5092 // file, then we found an include guard.
5093 if (IsEndIf && PP.IncludeGuard == IG_Defined && PP.BranchLevel == -1 &&
5094 getIncludeGuardState(Style.IndentPPDirectives) == IG_Inited &&
5095 (eof() ||
5096 (PreviousWasComment &&
5097 Tokens->peekNextToken(/*SkipComment=*/true)->is(tok::eof)))) {
5098 PP.IncludeGuard = IG_Found;
5099 }
5100 }
5101
5102 if (!PP.Stack.empty() && (PP.Stack.back().Kind == PP_Unreachable) &&
5103 !Line->InPPDirective) {
5104 continue;
5105 }
5106
5107 if (FormatTok->is(tok::identifier) &&
5108 Macros.defined(FormatTok->TokenText) &&
5109 // FIXME: Allow expanding macros in preprocessor directives.
5110 !Line->InPPDirective) {
5111 FormatToken *ID = FormatTok;
5112 unsigned Position = Tokens->getPosition();
5113 // Parsing the arguments of the call may parse preprocessor directives,
5114 // which are parsed again if the token stream is rewound because the
5115 // arguments are discarded. The preprocessor bookkeeping is restored
5116 // whenever that happens.
5117 const auto SavedPPState = PP;
5118
5119 // To correctly parse the code, we need to replace the tokens of the macro
5120 // call with its expansion.
5121 auto PreCall = std::move(Line);
5122 Line.reset(new UnwrappedLine);
5123 bool OldInExpansion = InExpansion;
5124 InExpansion = true;
5125 // We parse the macro call into a new line.
5126 auto Args = parseMacroCall(SavedPPState);
5127 InExpansion = OldInExpansion;
5128 assert(Line->Tokens.front().Tok == ID);
5129 // And remember the unexpanded macro call tokens.
5130 auto UnexpandedLine = std::move(Line);
5131 // Reset to the old line.
5132 Line = std::move(PreCall);
5133
5134 LLVM_DEBUG({
5135 llvm::dbgs() << "Macro call: " << ID->TokenText << "(";
5136 if (Args) {
5137 llvm::dbgs() << "(";
5138 for (const auto &Arg : Args.value())
5139 for (const auto &T : Arg)
5140 llvm::dbgs() << T->TokenText << " ";
5141 llvm::dbgs() << ")";
5142 }
5143 llvm::dbgs() << "\n";
5144 });
5145 if (Macros.objectLike(ID->TokenText) && Args &&
5146 !Macros.hasArity(ID->TokenText, Args->size())) {
5147 // The macro is either
5148 // - object-like, but we got argumnets, or
5149 // - overloaded to be both object-like and function-like, but none of
5150 // the function-like arities match the number of arguments.
5151 // Thus, expand as object-like macro.
5152 LLVM_DEBUG(llvm::dbgs()
5153 << "Macro \"" << ID->TokenText
5154 << "\" not overloaded for arity " << Args->size()
5155 << "or not function-like, using object-like overload.");
5156 Args.reset();
5157 UnexpandedLine->Tokens.resize(1);
5158 Tokens->setPosition(Position);
5159 // Not nextToken(), which would push the stale FormatTok onto the line.
5160 FormatTok = Tokens->getNextToken();
5161 PP = SavedPPState;
5162 assert(!Args && Macros.objectLike(ID->TokenText));
5163 }
5164 if ((!Args && Macros.objectLike(ID->TokenText)) ||
5165 (Args && Macros.hasArity(ID->TokenText, Args->size()))) {
5166 // Next, we insert the expanded tokens in the token stream at the
5167 // current position, and continue parsing.
5168 Unexpanded[ID] = std::move(UnexpandedLine);
5170 Macros.expand(ID, std::move(Args));
5171 if (!Expansion.empty())
5172 FormatTok = Tokens->insertTokens(Expansion);
5173
5174 LLVM_DEBUG({
5175 llvm::dbgs() << "Expanded: ";
5176 for (const auto &T : Expansion)
5177 llvm::dbgs() << T->TokenText << " ";
5178 llvm::dbgs() << "\n";
5179 });
5180 } else {
5181 LLVM_DEBUG({
5182 llvm::dbgs() << "Did not expand macro \"" << ID->TokenText
5183 << "\", because it was used ";
5184 if (Args)
5185 llvm::dbgs() << "with " << Args->size();
5186 else
5187 llvm::dbgs() << "without";
5188 llvm::dbgs() << " arguments, which doesn't match any definition.\n";
5189 });
5190 Tokens->setPosition(Position);
5191 FormatTok = ID;
5192 PP = SavedPPState;
5193 }
5194 }
5195
5196 if (FormatTok->isNot(tok::comment)) {
5197 distributeComments(Comments, FormatTok);
5198 Comments.clear();
5199 return;
5200 }
5201
5202 Comments.push_back(FormatTok);
5203 } while (!eof());
5204
5205 distributeComments(Comments, nullptr);
5206 Comments.clear();
5207}
5208
5209namespace {
5210template <typename Iterator>
5211void pushTokens(Iterator Begin, Iterator End,
5213 for (auto I = Begin; I != End; ++I) {
5214 Into.push_back(I->Tok);
5215 for (const auto &Child : I->Children)
5216 pushTokens(Child.Tokens.begin(), Child.Tokens.end(), Into);
5217 }
5218}
5219} // namespace
5220
5221std::optional<llvm::SmallVector<llvm::SmallVector<FormatToken *, 8>, 1>>
5222UnwrappedLineParser::parseMacroCall(const PPState &SavedPPState) {
5223 std::optional<llvm::SmallVector<llvm::SmallVector<FormatToken *, 8>, 1>> Args;
5224 assert(Line->Tokens.empty());
5225 // Not nextToken(), which would already expand a directly following macro
5226 // call before the expansion of this one is inserted.
5227 auto ConsumeLastTokenOfCall = [this] {
5228 flushComments(isOnNewLine(*FormatTok));
5229 pushToken(FormatTok);
5230 FormatTok = Tokens->getNextToken();
5231 };
5232 if (Tokens->peekNextToken(/*SkipComment=*/true)->isNot(tok::l_paren)) {
5233 ConsumeLastTokenOfCall();
5234 return Args;
5235 }
5236 nextToken();
5237 assert(FormatTok->is(tok::l_paren));
5238 unsigned Position = Tokens->getPosition();
5239 FormatToken *Tok = FormatTok;
5240 nextToken();
5241 Args.emplace();
5242 auto ArgStart = std::prev(Line->Tokens.end());
5243
5244 int Parens = 0;
5245 do {
5246 switch (FormatTok->Tok.getKind()) {
5247 case tok::l_paren:
5248 ++Parens;
5249 nextToken();
5250 break;
5251 case tok::r_paren: {
5252 if (Parens > 0) {
5253 --Parens;
5254 nextToken();
5255 break;
5256 }
5257 Args->push_back({});
5258 pushTokens(std::next(ArgStart), Line->Tokens.end(), Args->back());
5259 ConsumeLastTokenOfCall();
5260 return Args;
5261 }
5262 case tok::comma: {
5263 if (Parens > 0) {
5264 nextToken();
5265 break;
5266 }
5267 Args->push_back({});
5268 pushTokens(std::next(ArgStart), Line->Tokens.end(), Args->back());
5269 nextToken();
5270 ArgStart = std::prev(Line->Tokens.end());
5271 break;
5272 }
5273 default:
5274 nextToken();
5275 break;
5276 }
5277 } while (!eof());
5278 Line->Tokens.resize(1);
5279 Tokens->setPosition(Position);
5280 FormatTok = Tok;
5281 PP = SavedPPState;
5282 return {};
5283}
5284
5285void UnwrappedLineParser::pushToken(FormatToken *Tok) {
5286 Line->Tokens.push_back(UnwrappedLineNode(Tok));
5287 if (PP.AtEndOfPPLine) {
5288 auto &Tok = *Line->Tokens.back().Tok;
5289 Tok.MustBreakBefore = true;
5290 Tok.MustBreakBeforeFinalized = true;
5291 Tok.FirstAfterPPLine = true;
5292 PP.AtEndOfPPLine = false;
5293 }
5294}
5295
5296} // end namespace format
5297} // end namespace clang
This file defines the FormatTokenSource interface, which provides a token stream as well as the abili...
This file contains the declaration of the FormatToken, a wrapper around Token with additional informa...
FormatToken()
Token Tok
The Token.
unsigned OriginalColumn
The original 0-based column of this token, including expanded tabs.
FormatToken * Previous
The previous token in the unwrapped line.
FormatToken * Next
The next token in the unwrapped line.
This file contains the main building blocks of macro support in clang-format.
static bool HasAttribute(const QualType &T)
This file implements a token annotator, i.e.
Defines the clang::TokenKind enum and support functions.
This file contains the declaration of the UnwrappedLineParser, which turns a stream of tokens into Un...
Implements an efficient mapping from strings to IdentifierInfo nodes.
Parser - This implements a parser for the C family of languages.
Definition Parser.h:256
This class handles loading and caching of source files into memory.
Token - This structure provides full information about a lexed token.
Definition Token.h:36
IdentifierInfo * getIdentifierInfo() const
Definition Token.h:197
bool isLiteral() const
Return true if this is a "literal", like a numeric constant, string, etc.
Definition Token.h:126
bool is(tok::TokenKind K) const
is/isNot - Predicates to check if this token is a specific kind, as in "if (Tok.is(tok::l_brace)) {....
Definition Token.h:104
tok::TokenKind getKind() const
Definition Token.h:99
bool isOneOf(Ts... Ks) const
Definition Token.h:105
bool isNot(tok::TokenKind K) const
Definition Token.h:111
CompoundStatementIndenter(UnwrappedLineParser *Parser, const FormatStyle &Style, unsigned &LineLevel)
CompoundStatementIndenter(UnwrappedLineParser *Parser, unsigned &LineLevel, bool WrapBrace, bool IndentBrace)
ScopedLineState(UnwrappedLineParser &Parser, bool SwitchToPreprocessorLines=false, bool DiscardLines=false)
Interface for users of the UnwrappedLineParser to receive the parsed lines.
UnwrappedLineParser(SourceManager &SourceMgr, const FormatStyle &Style, const AdditionalKeywords &Keywords, unsigned FirstStartColumn, ArrayRef< FormatToken * > Tokens, UnwrappedLineConsumer &Callback, llvm::SpecificBumpPtrAllocator< FormatToken > &Allocator, IdentifierTable &IdentTable)
static void hash_combine(std::size_t &seed, const T &v)
static bool mustBeJSIdentOrValue(const AdditionalKeywords &Keywords, const FormatToken *FormatTok)
std::ostream & operator<<(std::ostream &Stream, const UnwrappedLine &Line)
static bool tokenCanStartNewLine(const FormatToken &Tok)
static bool continuesLineCommentSection(const FormatToken &FormatTok, const UnwrappedLine &Line, const FormatStyle &Style, const llvm::Regex &CommentPragmasRegex)
static bool isC78Type(const FormatToken &Tok)
static bool isJSDeclOrStmt(const AdditionalKeywords &Keywords, const FormatToken *FormatTok)
LangOptions getFormattingLangOpts(const FormatStyle &Style=getLLVMStyle())
Returns the LangOpts that the formatter expects you to set.
Definition Format.cpp:4559
static void markOptionalBraces(FormatToken *LeftBrace)
static bool mustBeJSIdent(const AdditionalKeywords &Keywords, const FormatToken *FormatTok)
static bool isIIFE(const UnwrappedLine &Line, const AdditionalKeywords &Keywords)
static bool isC78ParameterDecl(const FormatToken *Tok, const FormatToken *Next, const FormatToken *FuncName)
static bool isGoogScope(const UnwrappedLine &Line)
static FormatToken * getLastNonComment(const UnwrappedLine &Line)
TokenType
Determines the semantic type of a syntactic token, e.g.
static bool ShouldBreakBeforeBrace(const FormatStyle &Style, const FormatToken &InitialToken, bool IsEmptyBlock, bool IsJavaRecord=false)
TokenKind
Provides a simple uniform namespace for tokens from all C languages.
Definition TokenKinds.h:33
bool isLiteral(TokenKind K)
Return true if this is a "literal" kind, like a numeric constant, string, etc.
Definition TokenKinds.h:109
Top level wrappers for InstallAPI frontend operations.
bool isLineComment(const FormatToken &FormatTok)
if(T->getSizeExpr()) TRY_TO(TraverseStmt(const_cast< Expr * >(T -> getSizeExpr())))
nullptr
This class represents a compute construct, representing a 'Kind' of ‘parallel’, 'serial',...
@ Default
Set to the current date and time.
const FunctionProtoType * T
@ Type
The name was classified as a type.
Definition Sema.h:558
bool continuesLineComment(const FormatToken &FormatTok, const FormatToken *Previous, const FormatToken *MinColumnToken)
@ Parens
New-expression has a C++98 paren-delimited initializer.
Definition ExprCXX.h:2249
@ LK_C
Should be used for C.
Definition Format.h:3837
@ LK_Proto
Should be used for Protocol Buffers
Definition Format.h:3851
@ IEBS_AfterExternBlock
Backwards compatible with AfterExternBlock's indenting.
Definition Format.h:3285
@ IEBS_Indent
Indents extern blocks.
Definition Format.h:3299
@ PPDIS_BeforeHash
Indents directives before the hash.
Definition Format.h:3392
@ PPDIS_None
Does not indent any directives.
Definition Format.h:3374
@ LS_Cpp20
Parse and format as C++20.
Definition Format.h:5897
@ BWACS_Always
Always wrap braces after a control statement.
Definition Format.h:1415
@ BWACS_Never
Never wrap braces after a control statement.
Definition Format.h:1394
@ BS_Whitesmiths
Like Allman but always indent braces and line up code with braces.
Definition Format.h:2250
@ RPS_Leave
Do not remove parentheses.
Definition Format.h:4822
@ RPS_ReturnStatement
Also remove parentheses enclosing the expression in a return/co_return statement.
Definition Format.h:4837
@ NI_All
Indent in all namespaces.
Definition Format.h:4019
@ NI_Inner
Indent only in inner namespaces (nested in other namespaces).
Definition Format.h:4009
@ IGLS_OuterIndent
Indent goto labels to the enclosing block (previous indenting level).
Definition Format.h:3331
@ IGLS_NoIndent
Do not indent goto labels.
Definition Format.h:3319
Encapsulates keywords that are context sensitive or for languages not properly supported by Clang's l...
IdentifierInfo * kw_instanceof
IdentifierInfo * kw_implements
IdentifierInfo * kw_override
IdentifierInfo * kw_await
IdentifierInfo * kw_extends
IdentifierInfo * kw_async
IdentifierInfo * kw_from
IdentifierInfo * kw_abstract
IdentifierInfo * kw_var
IdentifierInfo * kw_interface
IdentifierInfo * kw_function
IdentifierInfo * kw_yield
IdentifierInfo * kw_where
IdentifierInfo * kw_throws
IdentifierInfo * kw_let
IdentifierInfo * kw_import
IdentifierInfo * kw_finally
Represents a complete lambda introducer.
Definition DeclSpec.h:2884
The FormatStyle is used to configure the formatting to follow specific guidelines.
Definition Format.h:51
@ LK_Proto
Should be used for Protocol Buffers
Definition Format.h:3851
@ RCS_Always
Apply indentation rules and reflow long comments into new lines, trying to obey the ColumnLimit.
Definition Format.h:4729
@ SRS_Empty
Only merge empty records.
Definition Format.h:1083
@ BS_Whitesmiths
Like Allman but always indent braces and line up code with braces.
Definition Format.h:2250
A wrapper around a Token storing information about the whitespace characters preceding it.
bool Optional
Is optional and can be removed.
bool isNot(T Kind) const
StringRef TokenText
The raw text of the token.
bool isNoneOf(Ts... Ks) const
unsigned NewlinesBefore
The number of newlines immediately before the Token.
bool is(tok::TokenKind Kind) const
bool isOneOf(A K1, B K2) const
unsigned IsFirst
Indicates that this is the first token of the file.
FormatToken * MatchingParen
If this is a bracket, this points to the matching one.
FormatToken * Previous
The previous token in the unwrapped line.
An unwrapped line is a sequence of Token, that we would like to put on a single line if there was no ...