clang 24.0.0git
UnwrappedLineParser.cpp
Go to the documentation of this file.
1//===--- UnwrappedLineParser.cpp - Format C++ code ------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8///
9/// \file
10/// This file contains the implementation of the UnwrappedLineParser,
11/// which turns a stream of tokens into UnwrappedLines.
12///
13//===----------------------------------------------------------------------===//
14
15#include "UnwrappedLineParser.h"
16#include "FormatToken.h"
17#include "FormatTokenSource.h"
18#include "Macros.h"
19#include "TokenAnnotator.h"
21#include "llvm/ADT/STLExtras.h"
22#include "llvm/ADT/StringRef.h"
23#include "llvm/Support/Debug.h"
24#include "llvm/Support/raw_os_ostream.h"
25#include "llvm/Support/raw_ostream.h"
26
27#include <utility>
28
29#define DEBUG_TYPE "format-parser"
30
31namespace clang {
32namespace format {
33
34namespace {
35
36void printLine(llvm::raw_ostream &OS, const UnwrappedLine &Line,
37 StringRef Prefix = "", bool PrintText = false) {
38 OS << Prefix << "Line(" << Line.Level << ", FSC=" << Line.FirstStartColumn
39 << ")" << (Line.InPPDirective ? " MACRO" : "") << ": ";
40 bool NewLine = false;
41 for (std::list<UnwrappedLineNode>::const_iterator I = Line.Tokens.begin(),
42 E = Line.Tokens.end();
43 I != E; ++I) {
44 if (NewLine) {
45 OS << Prefix;
46 NewLine = false;
47 }
48 OS << I->Tok->Tok.getName() << "["
49 << "T=" << (unsigned)I->Tok->getType()
50 << ", OC=" << I->Tok->OriginalColumn << ", \"" << I->Tok->TokenText
51 << "\"] ";
52 for (const auto *CI = I->Children.begin(), *CE = I->Children.end();
53 CI != CE; ++CI) {
54 OS << "\n";
55 printLine(OS, *CI, (Prefix + " ").str());
56 NewLine = true;
57 }
58 }
59 if (!NewLine)
60 OS << "\n";
61}
62
63[[maybe_unused]] static void printDebugInfo(const UnwrappedLine &Line) {
64 printLine(llvm::dbgs(), Line);
65}
66
67class ScopedDeclarationState {
68public:
69 ScopedDeclarationState(UnwrappedLine &Line, llvm::BitVector &Stack,
70 bool MustBeDeclaration)
71 : Line(Line), Stack(Stack) {
72 Line.MustBeDeclaration = MustBeDeclaration;
73 Stack.push_back(MustBeDeclaration);
74 }
75 ~ScopedDeclarationState() {
76 Stack.pop_back();
77 if (!Stack.empty())
78 Line.MustBeDeclaration = Stack.back();
79 else
80 Line.MustBeDeclaration = true;
81 }
82
83private:
84 UnwrappedLine &Line;
85 llvm::BitVector &Stack;
86};
87
88} // end anonymous namespace
89
90std::ostream &operator<<(std::ostream &Stream, const UnwrappedLine &Line) {
91 llvm::raw_os_ostream OS(Stream);
92 printLine(OS, Line);
93 return Stream;
94}
95
97public:
99 bool SwitchToPreprocessorLines = false)
100 : Parser(Parser), OriginalLines(Parser.CurrentLines) {
101 if (SwitchToPreprocessorLines)
102 Parser.CurrentLines = &Parser.PreprocessorDirectives;
103 else if (!Parser.Line->Tokens.empty())
104 Parser.CurrentLines = &Parser.Line->Tokens.back().Children;
105 PreBlockLine = std::move(Parser.Line);
106 Parser.Line = std::make_unique<UnwrappedLine>();
107 Parser.Line->Level = PreBlockLine->Level;
108 Parser.Line->PPLevel = PreBlockLine->PPLevel;
109 Parser.Line->InPPDirective = PreBlockLine->InPPDirective;
110 Parser.Line->InMacroBody = PreBlockLine->InMacroBody;
111 Parser.Line->UnbracedBodyLevel = PreBlockLine->UnbracedBodyLevel;
112 }
113
115 if (!Parser.Line->Tokens.empty())
116 Parser.addUnwrappedLine();
117 assert(Parser.Line->Tokens.empty());
118 Parser.Line = std::move(PreBlockLine);
119 if (Parser.CurrentLines == &Parser.PreprocessorDirectives)
120 Parser.AtEndOfPPLine = true;
121 Parser.CurrentLines = OriginalLines;
122 }
123
124private:
126
127 std::unique_ptr<UnwrappedLine> PreBlockLine;
128 SmallVectorImpl<UnwrappedLine> *OriginalLines;
129};
130
132public:
134 const FormatStyle &Style, unsigned &LineLevel)
136 Style.BraceWrapping.AfterControlStatement ==
137 FormatStyle::BWACS_Always,
138 Style.BraceWrapping.IndentBraces) {}
140 bool WrapBrace, bool IndentBrace)
141 : LineLevel(LineLevel), OldLineLevel(LineLevel) {
142 if (WrapBrace)
143 Parser->addUnwrappedLine();
144 if (IndentBrace)
145 ++LineLevel;
146 }
147 ~CompoundStatementIndenter() { LineLevel = OldLineLevel; }
148
149private:
150 unsigned &LineLevel;
151 unsigned OldLineLevel;
152};
153
155 SourceManager &SourceMgr, const FormatStyle &Style,
156 const AdditionalKeywords &Keywords, unsigned FirstStartColumn,
158 llvm::SpecificBumpPtrAllocator<FormatToken> &Allocator,
159 IdentifierTable &IdentTable)
160 : Line(new UnwrappedLine), AtEndOfPPLine(false), CurrentLines(&Lines),
161 Style(Style), IsCpp(Style.isCpp()),
162 LangOpts(getFormattingLangOpts(Style)), Keywords(Keywords),
163 CommentPragmasRegex(Style.CommentPragmas), Tokens(nullptr),
164 Callback(Callback), AllTokens(Tokens), PPBranchLevel(-1),
165 IncludeGuard(getIncludeGuardState(Style.IndentPPDirectives)),
166 IncludeGuardToken(nullptr), FirstStartColumn(FirstStartColumn),
167 Macros(Style.Macros, SourceMgr, Style, Allocator, IdentTable) {}
168
169void UnwrappedLineParser::reset() {
170 PPBranchLevel = -1;
171 IncludeGuard = getIncludeGuardState(Style.IndentPPDirectives);
172 IncludeGuardToken = nullptr;
173 Line.reset(new UnwrappedLine);
174 CommentsBeforeNextToken.clear();
175 FormatTok = nullptr;
176 AtEndOfPPLine = false;
177 IsDecltypeAutoFunction = false;
178 PreprocessorDirectives.clear();
179 CurrentLines = &Lines;
180 DeclarationScopeStack.clear();
181 NestedTooDeep.clear();
182 NestedLambdas.clear();
183 PPStack.clear();
184 Line->FirstStartColumn = FirstStartColumn;
185
186 if (!Unexpanded.empty())
187 for (FormatToken *Token : AllTokens)
188 Token->MacroCtx.reset();
189 CurrentExpandedLines.clear();
190 ExpandedLines.clear();
191 Unexpanded.clear();
192 InExpansion = false;
193 Reconstruct.reset();
194}
195
197 IndexedTokenSource TokenSource(AllTokens);
198 Line->FirstStartColumn = FirstStartColumn;
199 do {
200 LLVM_DEBUG(llvm::dbgs() << "----\n");
201 reset();
202 Tokens = &TokenSource;
203 TokenSource.reset();
204
205 readToken();
206 parseFile();
207
208 // If we found an include guard then all preprocessor directives (other than
209 // the guard) are over-indented by one.
210 if (IncludeGuard == IG_Found) {
211 for (auto &Line : Lines)
212 if (Line.InPPDirective && Line.Level > 0)
213 --Line.Level;
214 }
215
216 // Create line with eof token.
217 assert(eof());
218 pushToken(FormatTok);
219 addUnwrappedLine();
220
221 // In a first run, format everything with the lines containing macro calls
222 // replaced by the expansion.
223 if (!ExpandedLines.empty()) {
224 LLVM_DEBUG(llvm::dbgs() << "Expanded lines:\n");
225 for (const auto &Line : Lines) {
226 if (!Line.Tokens.empty()) {
227 auto it = ExpandedLines.find(Line.Tokens.begin()->Tok);
228 if (it != ExpandedLines.end()) {
229 for (const auto &Expanded : it->second) {
230 LLVM_DEBUG(printDebugInfo(Expanded));
231 Callback.consumeUnwrappedLine(Expanded);
232 }
233 continue;
234 }
235 }
236 LLVM_DEBUG(printDebugInfo(Line));
237 Callback.consumeUnwrappedLine(Line);
238 }
239 Callback.finishRun();
240 }
241
242 LLVM_DEBUG(llvm::dbgs() << "Unwrapped lines:\n");
243 for (const UnwrappedLine &Line : Lines) {
244 LLVM_DEBUG(printDebugInfo(Line));
245 Callback.consumeUnwrappedLine(Line);
246 }
247 Callback.finishRun();
248 Lines.clear();
249 while (!PPLevelBranchIndex.empty() &&
250 PPLevelBranchIndex.back() + 1 >= PPLevelBranchCount.back()) {
251 PPLevelBranchIndex.resize(PPLevelBranchIndex.size() - 1);
252 PPLevelBranchCount.resize(PPLevelBranchCount.size() - 1);
253 }
254 if (!PPLevelBranchIndex.empty()) {
255 ++PPLevelBranchIndex.back();
256 assert(PPLevelBranchIndex.size() == PPLevelBranchCount.size());
257 assert(PPLevelBranchIndex.back() <= PPLevelBranchCount.back());
258 }
259 } while (!PPLevelBranchIndex.empty());
260}
261
262void UnwrappedLineParser::parseFile() {
263 // The top-level context in a file always has declarations, except for pre-
264 // processor directives and JavaScript files.
265 bool MustBeDeclaration = !Line->InPPDirective && !Style.isJavaScript();
266 ScopedDeclarationState DeclarationState(*Line, DeclarationScopeStack,
267 MustBeDeclaration);
268 if (Style.isTextProto() || (Style.isJson() && FormatTok->IsFirst))
269 parseBracedList();
270 else
271 parseLevel();
272 // Make sure to format the remaining tokens.
273 //
274 // LK_TextProto is special since its top-level is parsed as the body of a
275 // braced list, which does not necessarily have natural line separators such
276 // as a semicolon. Comments after the last entry that have been determined to
277 // not belong to that line, as in:
278 // key: value
279 // // endfile comment
280 // do not have a chance to be put on a line of their own until this point.
281 // Here we add this newline before end-of-file comments.
282 if (Style.isTextProto() && !CommentsBeforeNextToken.empty())
283 addUnwrappedLine();
284 flushComments(true);
285 addUnwrappedLine();
286}
287
288void UnwrappedLineParser::parseCSharpGenericTypeConstraint() {
289 do {
290 switch (FormatTok->Tok.getKind()) {
291 case tok::l_brace:
292 case tok::semi:
293 return;
294 default:
295 if (FormatTok->is(Keywords.kw_where)) {
296 addUnwrappedLine();
297 nextToken();
298 parseCSharpGenericTypeConstraint();
299 break;
300 }
301 nextToken();
302 break;
303 }
304 } while (!eof());
305}
306
307void UnwrappedLineParser::parseCSharpAttribute() {
308 int UnpairedSquareBrackets = 1;
309 do {
310 switch (FormatTok->Tok.getKind()) {
311 case tok::r_square:
312 nextToken();
313 --UnpairedSquareBrackets;
314 if (UnpairedSquareBrackets == 0) {
315 addUnwrappedLine();
316 return;
317 }
318 break;
319 case tok::l_square:
320 ++UnpairedSquareBrackets;
321 nextToken();
322 break;
323 default:
324 nextToken();
325 break;
326 }
327 } while (!eof());
328}
329
330bool UnwrappedLineParser::precededByCommentOrPPDirective() const {
331 if (!Lines.empty() && Lines.back().InPPDirective)
332 return true;
333
334 const FormatToken *Previous = Tokens->getPreviousToken();
335 return Previous && Previous->is(tok::comment) &&
336 (Previous->IsMultiline || Previous->NewlinesBefore > 0);
337}
338
339/// Parses a level, that is ???.
340/// \param OpeningBrace Opening brace (\p nullptr if absent) of that level.
341/// \param IfKind The \p if statement kind in the level.
342/// \param IfLeftBrace The left brace of the \p if block in the level.
343/// \returns true if a simple block of if/else/for/while, or false otherwise.
344/// (A simple block has a single statement.)
345bool UnwrappedLineParser::parseLevel(const FormatToken *OpeningBrace,
346 IfStmtKind *IfKind,
347 FormatToken **IfLeftBrace) {
348 const bool InRequiresExpression =
349 OpeningBrace && OpeningBrace->is(TT_RequiresExpressionLBrace);
350 const bool IsPrecededByCommentOrPPDirective =
351 !Style.RemoveBracesLLVM || precededByCommentOrPPDirective();
352 FormatToken *IfLBrace = nullptr;
353 bool HasDoWhile = false;
354 bool HasLabel = false;
355 unsigned StatementCount = 0;
356 bool SwitchLabelEncountered = false;
357
358 do {
359 if (FormatTok->isAttribute()) {
360 nextToken();
361 if (FormatTok->is(tok::l_paren))
362 parseParens();
363 continue;
364 }
365 tok::TokenKind Kind = FormatTok->Tok.getKind();
366 if (FormatTok->is(TT_MacroBlockBegin))
367 Kind = tok::l_brace;
368 else if (FormatTok->is(TT_MacroBlockEnd))
369 Kind = tok::r_brace;
370
371 auto ParseDefault = [this, OpeningBrace, IfKind, &IfLBrace, &HasDoWhile,
372 &HasLabel, &StatementCount] {
373 parseStructuralElement(OpeningBrace, IfKind, &IfLBrace,
374 HasDoWhile ? nullptr : &HasDoWhile,
375 HasLabel ? nullptr : &HasLabel);
376 ++StatementCount;
377 assert(StatementCount > 0 && "StatementCount overflow!");
378 };
379
380 switch (Kind) {
381 case tok::comment:
382 nextToken();
383 addUnwrappedLine();
384 break;
385 case tok::l_brace:
386 if (InRequiresExpression) {
387 FormatTok->setFinalizedType(TT_CompoundRequirementLBrace);
388 } else if (FormatTok->Previous &&
389 FormatTok->Previous->ClosesRequiresClause) {
390 // We need the 'default' case here to correctly parse a function
391 // l_brace.
392 ParseDefault();
393 continue;
394 }
395 if (!InRequiresExpression && FormatTok->isNot(TT_MacroBlockBegin)) {
396 if (tryToParseBracedList())
397 continue;
398 FormatTok->setFinalizedType(TT_BlockLBrace);
399 }
400 parseBlock();
401 ++StatementCount;
402 assert(StatementCount > 0 && "StatementCount overflow!");
403 addUnwrappedLine();
404 break;
405 case tok::r_brace:
406 if (OpeningBrace) {
407 if (!Style.RemoveBracesLLVM || Line->InPPDirective ||
408 OpeningBrace->isNoneOf(TT_ControlStatementLBrace, TT_ElseLBrace)) {
409 return false;
410 }
411 if (FormatTok->isNot(tok::r_brace) || StatementCount != 1 || HasLabel ||
412 HasDoWhile || IsPrecededByCommentOrPPDirective ||
413 precededByCommentOrPPDirective()) {
414 return false;
415 }
416 const FormatToken *Next = Tokens->peekNextToken();
417 if (Next->is(tok::comment) && Next->NewlinesBefore == 0)
418 return false;
419 if (IfLeftBrace)
420 *IfLeftBrace = IfLBrace;
421 return true;
422 }
423 nextToken();
424 addUnwrappedLine();
425 break;
426 case tok::kw_default: {
427 unsigned StoredPosition = Tokens->getPosition();
428 auto *Next = Tokens->getNextNonComment();
429 FormatTok = Tokens->setPosition(StoredPosition);
430 if (Next->isNoneOf(tok::colon, tok::arrow)) {
431 // default not followed by `:` or `->` is not a case label; treat it
432 // like an identifier.
433 parseStructuralElement();
434 break;
435 }
436 // Else, if it is 'default:', fall through to the case handling.
437 [[fallthrough]];
438 }
439 case tok::kw_case:
440 if (Style.Language == FormatStyle::LK_Proto || Style.isVerilog() ||
441 (Style.isJavaScript() && Line->MustBeDeclaration)) {
442 // Proto: there are no switch/case statements
443 // Verilog: Case labels don't have this word. We handle case
444 // labels including default in TokenAnnotator.
445 // JavaScript: A 'case: string' style field declaration.
446 ParseDefault();
447 break;
448 }
449 if (!SwitchLabelEncountered &&
450 (Style.IndentCaseLabels ||
451 (OpeningBrace && OpeningBrace->is(TT_SwitchExpressionLBrace)) ||
452 (Line->InPPDirective && Line->Level == 1))) {
453 ++Line->Level;
454 }
455 SwitchLabelEncountered = true;
456 parseStructuralElement();
457 break;
458 case tok::l_square:
459 if (Style.isCSharp()) {
460 nextToken();
461 parseCSharpAttribute();
462 break;
463 }
464 if (handleCppAttributes())
465 break;
466 [[fallthrough]];
467 default:
468 ParseDefault();
469 break;
470 }
471 } while (!eof());
472
473 return false;
474}
475
476void UnwrappedLineParser::calculateBraceTypes(bool ExpectClassBody) {
477 // We'll parse forward through the tokens until we hit
478 // a closing brace or eof - note that getNextToken() will
479 // parse macros, so this will magically work inside macro
480 // definitions, too.
481 unsigned StoredPosition = Tokens->getPosition();
482 FormatToken *Tok = FormatTok;
483 const FormatToken *PrevTok = Tok->Previous;
484 // Keep a stack of positions of lbrace tokens. We will
485 // update information about whether an lbrace starts a
486 // braced init list or a different block during the loop.
487 struct StackEntry {
489 const FormatToken *PrevTok;
490 };
491 SmallVector<StackEntry, 8> LBraceStack;
492 assert(Tok->is(tok::l_brace));
493
494 do {
495 auto *NextTok = Tokens->getNextNonComment();
496
497 if (!Line->InMacroBody && !Style.isTableGen()) {
498 // Skip PPDirective lines (except macro definitions) and comments.
499 while (NextTok->is(tok::hash)) {
500 NextTok = Tokens->getNextToken();
501 if (NextTok->isOneOf(tok::pp_not_keyword, tok::pp_define))
502 break;
503 do {
504 NextTok = Tokens->getNextToken();
505 } while (!NextTok->HasUnescapedNewline && NextTok->isNot(tok::eof));
506
507 while (NextTok->is(tok::comment))
508 NextTok = Tokens->getNextToken();
509 }
510 }
511
512 switch (Tok->Tok.getKind()) {
513 case tok::l_brace:
514 if (Style.isJavaScript() && PrevTok) {
515 if (PrevTok->isOneOf(tok::colon, tok::less)) {
516 // A ':' indicates this code is in a type, or a braced list
517 // following a label in an object literal ({a: {b: 1}}).
518 // A '<' could be an object used in a comparison, but that is nonsense
519 // code (can never return true), so more likely it is a generic type
520 // argument (`X<{a: string; b: number}>`).
521 // The code below could be confused by semicolons between the
522 // individual members in a type member list, which would normally
523 // trigger BK_Block. In both cases, this must be parsed as an inline
524 // braced init.
525 Tok->setBlockKind(BK_BracedInit);
526 } else if (PrevTok->is(tok::r_paren)) {
527 // `) { }` can only occur in function or method declarations in JS.
528 Tok->setBlockKind(BK_Block);
529 }
530 } else if (Style.isJava() && PrevTok && PrevTok->is(tok::arrow)) {
531 Tok->setBlockKind(BK_Block);
532 } else {
533 Tok->setBlockKind(BK_Unknown);
534 }
535 LBraceStack.push_back({Tok, PrevTok});
536 break;
537 case tok::r_brace:
538 if (LBraceStack.empty())
539 break;
540 if (auto *LBrace = LBraceStack.back().Tok; LBrace->is(BK_Unknown)) {
541 bool ProbablyBracedList = false;
542 if (Style.Language == FormatStyle::LK_Proto) {
543 ProbablyBracedList = NextTok->isOneOf(tok::comma, tok::r_square);
544 } else if (LBrace->isNot(TT_EnumLBrace)) {
545 // Using OriginalColumn to distinguish between ObjC methods and
546 // binary operators is a bit hacky.
547 bool NextIsObjCMethod = NextTok->isOneOf(tok::plus, tok::minus) &&
548 NextTok->OriginalColumn == 0;
549
550 // Try to detect a braced list. Note that regardless how we mark inner
551 // braces here, we will overwrite the BlockKind later if we parse a
552 // braced list (where all blocks inside are by default braced lists),
553 // or when we explicitly detect blocks (for example while parsing
554 // lambdas).
555
556 // If we already marked the opening brace as braced list, the closing
557 // must also be part of it.
558 ProbablyBracedList = LBrace->is(TT_BracedListLBrace);
559
560 ProbablyBracedList = ProbablyBracedList ||
561 (Style.isJavaScript() &&
562 NextTok->isOneOf(Keywords.kw_of, Keywords.kw_in,
563 Keywords.kw_as));
564 ProbablyBracedList =
565 ProbablyBracedList ||
566 (IsCpp && (PrevTok->Tok.isLiteral() ||
567 NextTok->isOneOf(tok::l_paren, tok::arrow)));
568
569 // If there is a comma, or right paren after the closing brace, we
570 // assume this is a braced initializer list.
571 // FIXME: Some of these do not apply to JS, e.g. "} {" can never be a
572 // braced list in JS.
573 ProbablyBracedList =
574 ProbablyBracedList ||
575 NextTok->isOneOf(tok::comma, tok::period, tok::colon,
576 tok::r_paren, tok::r_square, tok::ellipsis);
577
578 // Distinguish between braced list in a constructor initializer list
579 // followed by constructor body, or just adjacent blocks.
580 ProbablyBracedList =
581 ProbablyBracedList ||
582 (NextTok->is(tok::l_brace) && LBraceStack.back().PrevTok &&
583 LBraceStack.back().PrevTok->isOneOf(tok::identifier,
584 tok::greater));
585
586 ProbablyBracedList =
587 ProbablyBracedList ||
588 (NextTok->is(tok::identifier) &&
589 PrevTok->isNoneOf(tok::semi, tok::r_brace, tok::l_brace));
590
591 ProbablyBracedList = ProbablyBracedList ||
592 (NextTok->is(tok::semi) &&
593 (!ExpectClassBody || LBraceStack.size() != 1));
594
595 ProbablyBracedList =
596 ProbablyBracedList ||
597 (NextTok->isBinaryOperator() && !NextIsObjCMethod);
598
599 if (!Style.isCSharp() && NextTok->is(tok::l_square)) {
600 // We can have an array subscript after a braced init
601 // list, but C++11 attributes are expected after blocks.
602 NextTok = Tokens->getNextToken();
603 ProbablyBracedList = NextTok->isNot(tok::l_square);
604 }
605
606 // Cpp macro definition body that is a nonempty braced list or block:
607 if (IsCpp && Line->InMacroBody && PrevTok != FormatTok &&
608 !FormatTok->Previous && NextTok->is(tok::eof) &&
609 // A statement can end with only `;` (simple statement), a block
610 // closing brace (compound statement), or `:` (label statement).
611 // If PrevTok is a block opening brace, Tok ends an empty block.
612 PrevTok->isNoneOf(tok::semi, BK_Block, tok::colon)) {
613 ProbablyBracedList = true;
614 }
615 }
616 const auto BlockKind = ProbablyBracedList ? BK_BracedInit : BK_Block;
617 Tok->setBlockKind(BlockKind);
618 LBrace->setBlockKind(BlockKind);
619 }
620 LBraceStack.pop_back();
621 break;
622 case tok::identifier:
623 if (Tok->isNot(TT_StatementMacro))
624 break;
625 [[fallthrough]];
626 case tok::at:
627 case tok::semi:
628 case tok::kw_if:
629 case tok::kw_while:
630 case tok::kw_for:
631 case tok::kw_switch:
632 case tok::kw_try:
633 case tok::kw___try:
634 if (!LBraceStack.empty() && LBraceStack.back().Tok->is(BK_Unknown))
635 LBraceStack.back().Tok->setBlockKind(BK_Block);
636 break;
637 default:
638 break;
639 }
640
641 PrevTok = Tok;
642 Tok = NextTok;
643 } while (Tok->isNot(tok::eof) && !LBraceStack.empty());
644
645 // Assume other blocks for all unclosed opening braces.
646 for (const auto &Entry : LBraceStack)
647 if (Entry.Tok->is(BK_Unknown))
648 Entry.Tok->setBlockKind(BK_Block);
649
650 FormatTok = Tokens->setPosition(StoredPosition);
651}
652
653// Sets the token type of the directly previous right brace.
654void UnwrappedLineParser::setPreviousRBraceType(TokenType Type) {
655 if (auto Prev = FormatTok->getPreviousNonComment();
656 Prev && Prev->is(tok::r_brace)) {
657 Prev->setFinalizedType(Type);
658 }
659}
660
661template <class T>
662static inline void hash_combine(std::size_t &seed, const T &v) {
663 std::hash<T> hasher;
664 seed ^= hasher(v) + 0x9e3779b9 + (seed << 6) + (seed >> 2);
665}
666
667size_t UnwrappedLineParser::computePPHash() const {
668 size_t h = 0;
669 for (const auto &i : PPStack) {
670 hash_combine(h, size_t(i.Kind));
671 hash_combine(h, i.Line);
672 }
673 return h;
674}
675
676// Checks whether \p ParsedLine might fit on a single line. If \p OpeningBrace
677// is not null, subtracts its length (plus the preceding space) when computing
678// the length of \p ParsedLine. We must clone the tokens of \p ParsedLine before
679// running the token annotator on it so that we can restore them afterward.
680bool UnwrappedLineParser::mightFitOnOneLine(
681 UnwrappedLine &ParsedLine, const FormatToken *OpeningBrace) const {
682 const auto ColumnLimit = Style.ColumnLimit;
683 if (ColumnLimit == 0)
684 return true;
685
686 auto &Tokens = ParsedLine.Tokens;
687 assert(!Tokens.empty());
688
689 const auto *LastToken = Tokens.back().Tok;
690 assert(LastToken);
691
692 SmallVector<UnwrappedLineNode> SavedTokens(Tokens.size());
693
694 int Index = 0;
695 for (const auto &Token : Tokens) {
696 assert(Token.Tok);
697 auto &SavedToken = SavedTokens[Index++];
698 SavedToken.Tok = new FormatToken;
699 SavedToken.Tok->copyFrom(*Token.Tok);
700 SavedToken.Children = std::move(Token.Children);
701 }
702
703 AnnotatedLine Line(ParsedLine);
704 assert(Line.Last == LastToken);
705
706 TokenAnnotator Annotator(Style, Keywords);
707 Annotator.annotate(Line);
708 Annotator.calculateFormattingInformation(Line);
709
710 auto Length = LastToken->TotalLength;
711 if (OpeningBrace) {
712 assert(OpeningBrace != Tokens.front().Tok);
713 if (auto Prev = OpeningBrace->Previous;
714 Prev && Prev->TotalLength + ColumnLimit == OpeningBrace->TotalLength) {
715 Length -= ColumnLimit;
716 }
717 Length -= OpeningBrace->TokenText.size() + 1;
718 }
719
720 if (const auto *FirstToken = Line.First; FirstToken->is(tok::r_brace)) {
721 assert(!OpeningBrace || OpeningBrace->is(TT_ControlStatementLBrace));
722 Length -= FirstToken->TokenText.size() + 1;
723 }
724
725 Index = 0;
726 for (auto &Token : Tokens) {
727 const auto &SavedToken = SavedTokens[Index++];
728 Token.Tok->copyFrom(*SavedToken.Tok);
729 Token.Children = std::move(SavedToken.Children);
730 delete SavedToken.Tok;
731 }
732
733 // If these change PPLevel needs to be used for get correct indentation.
734 assert(!Line.InMacroBody);
735 assert(!Line.InPPDirective);
736 return Line.Level * Style.IndentWidth + Length <= ColumnLimit;
737}
738
739FormatToken *UnwrappedLineParser::parseBlock(bool MustBeDeclaration,
740 unsigned AddLevels, bool MunchSemi,
741 bool KeepBraces,
742 IfStmtKind *IfKind,
743 bool UnindentWhitesmithsBraces) {
744 auto HandleVerilogBlockLabel = [this]() {
745 // ":" name
746 if (Style.isVerilog() && FormatTok->is(tok::colon)) {
747 nextToken();
748 if (Keywords.isVerilogIdentifier(*FormatTok))
749 nextToken();
750 }
751 };
752
753 // Whether this is a Verilog-specific block that has a special header like a
754 // module.
755 const bool VerilogHierarchy =
756 Style.isVerilog() && Keywords.isVerilogHierarchy(*FormatTok);
757 assert((FormatTok->isOneOf(tok::l_brace, TT_MacroBlockBegin) ||
758 (Style.isVerilog() &&
759 (Keywords.isVerilogBegin(*FormatTok) || VerilogHierarchy))) &&
760 "'{' or macro block token expected");
761 FormatToken *Tok = FormatTok;
762 const bool FollowedByComment = Tokens->peekNextToken()->is(tok::comment);
763 auto Index = CurrentLines->size();
764 const bool MacroBlock = FormatTok->is(TT_MacroBlockBegin);
765 FormatTok->setBlockKind(BK_Block);
766
767 const bool IsWhitesmiths =
768 Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths;
769
770 // For Whitesmiths mode, jump to the next level prior to skipping over the
771 // braces.
772 if (!VerilogHierarchy && AddLevels > 0 && IsWhitesmiths)
773 ++Line->Level;
774
775 size_t PPStartHash = computePPHash();
776
777 const unsigned InitialLevel = Line->Level;
778 if (VerilogHierarchy) {
779 AddLevels += parseVerilogHierarchyHeader();
780 } else {
781 nextToken(/*LevelDifference=*/AddLevels);
782 HandleVerilogBlockLabel();
783 }
784
785 // Bail out if there are too many levels. Otherwise, the stack might overflow.
786 if (Line->Level > 300)
787 return nullptr;
788
789 if (MacroBlock && FormatTok->is(tok::l_paren))
790 parseParens();
791
792 size_t NbPreprocessorDirectives =
793 !parsingPPDirective() ? PreprocessorDirectives.size() : 0;
794 addUnwrappedLine();
795 size_t OpeningLineIndex =
796 CurrentLines->empty()
798 : (CurrentLines->size() - 1 - NbPreprocessorDirectives);
799
800 // Whitesmiths is weird here. The brace needs to be indented for the namespace
801 // block, but the block itself may not be indented depending on the style
802 // settings. This allows the format to back up one level in those cases.
803 if (UnindentWhitesmithsBraces)
804 --Line->Level;
805
806 ScopedDeclarationState DeclarationState(*Line, DeclarationScopeStack,
807 MustBeDeclaration);
808
809 // Whitesmiths logic has already added a level by this point, so avoid
810 // adding it twice.
811 if (AddLevels > 0u)
812 Line->Level += AddLevels - (IsWhitesmiths ? 1 : 0);
813
814 FormatToken *IfLBrace = nullptr;
815 const bool SimpleBlock = parseLevel(Tok, IfKind, &IfLBrace);
816
817 if (eof())
818 return IfLBrace;
819
820 if (MacroBlock ? FormatTok->isNot(TT_MacroBlockEnd)
821 : FormatTok->isNot(tok::r_brace)) {
822 Line->Level = InitialLevel;
823 FormatTok->setBlockKind(BK_Block);
824 return IfLBrace;
825 }
826
827 if (FormatTok->is(tok::r_brace)) {
828 FormatTok->setBlockKind(BK_Block);
829 if (Tok->is(TT_NamespaceLBrace))
830 FormatTok->setFinalizedType(TT_NamespaceRBrace);
831 }
832
833 const bool IsFunctionRBrace =
834 FormatTok->is(tok::r_brace) && Tok->is(TT_FunctionLBrace);
835
836 auto RemoveBraces = [=]() mutable {
837 if (!SimpleBlock)
838 return false;
839 assert(Tok->isOneOf(TT_ControlStatementLBrace, TT_ElseLBrace));
840 assert(FormatTok->is(tok::r_brace));
841 const bool WrappedOpeningBrace = !Tok->Previous;
842 if (WrappedOpeningBrace && FollowedByComment)
843 return false;
844 const bool HasRequiredIfBraces = IfLBrace && !IfLBrace->Optional;
845 if (KeepBraces && !HasRequiredIfBraces)
846 return false;
847 if (Tok->isNot(TT_ElseLBrace) || !HasRequiredIfBraces) {
848 const FormatToken *Previous = Tokens->getPreviousToken();
849 assert(Previous);
850 if (Previous->is(tok::r_brace) && !Previous->Optional)
851 return false;
852 }
853 assert(!CurrentLines->empty());
854 auto &LastLine = CurrentLines->back();
855 if (LastLine.Level == InitialLevel + 1 && !mightFitOnOneLine(LastLine))
856 return false;
857 if (Tok->is(TT_ElseLBrace))
858 return true;
859 if (WrappedOpeningBrace) {
860 assert(Index > 0);
861 --Index; // The line above the wrapped l_brace.
862 Tok = nullptr;
863 }
864 return mightFitOnOneLine((*CurrentLines)[Index], Tok);
865 };
866 if (RemoveBraces()) {
867 Tok->MatchingParen = FormatTok;
868 FormatTok->MatchingParen = Tok;
869 }
870
871 size_t PPEndHash = computePPHash();
872
873 // Munch the closing brace.
874 nextToken(/*LevelDifference=*/-AddLevels);
875
876 // When this is a function block and there is an unnecessary semicolon
877 // afterwards then mark it as optional (so the RemoveSemi pass can get rid of
878 // it later).
879 if (Style.RemoveSemicolon && IsFunctionRBrace) {
880 while (FormatTok->is(tok::semi)) {
881 FormatTok->Optional = true;
882 nextToken();
883 }
884 }
885
886 HandleVerilogBlockLabel();
887
888 if (MacroBlock && FormatTok->is(tok::l_paren))
889 parseParens();
890
891 Line->Level = InitialLevel;
892
893 if (FormatTok->is(tok::kw_noexcept)) {
894 // A noexcept in a requires expression.
895 nextToken();
896 }
897
898 if (FormatTok->is(tok::arrow)) {
899 // Following the } or noexcept we can find a trailing return type arrow
900 // as part of an implicit conversion constraint.
901 nextToken();
902 parseStructuralElement();
903 }
904
905 if (MunchSemi && FormatTok->is(tok::semi))
906 nextToken();
907
908 if (PPStartHash == PPEndHash) {
909 Line->MatchingOpeningBlockLineIndex = OpeningLineIndex;
910 if (OpeningLineIndex != UnwrappedLine::kInvalidIndex) {
911 // Update the opening line to add the forward reference as well
912 (*CurrentLines)[OpeningLineIndex].MatchingClosingBlockLineIndex =
913 CurrentLines->size() - 1;
914 }
915 }
916
917 return IfLBrace;
918}
919
920static bool isGoogScope(const UnwrappedLine &Line) {
921 // FIXME: Closure-library specific stuff should not be hard-coded but be
922 // configurable.
923 if (Line.Tokens.size() < 4)
924 return false;
925 auto I = Line.Tokens.begin();
926 if (I->Tok->TokenText != "goog")
927 return false;
928 ++I;
929 if (I->Tok->isNot(tok::period))
930 return false;
931 ++I;
932 if (I->Tok->TokenText != "scope")
933 return false;
934 ++I;
935 return I->Tok->is(tok::l_paren);
936}
937
938static bool isIIFE(const UnwrappedLine &Line,
939 const AdditionalKeywords &Keywords) {
940 // Look for the start of an immediately invoked anonymous function.
941 // https://en.wikipedia.org/wiki/Immediately-invoked_function_expression
942 // This is commonly done in JavaScript to create a new, anonymous scope.
943 // Example: (function() { ... })()
944 if (Line.Tokens.size() < 3)
945 return false;
946 auto I = Line.Tokens.begin();
947 if (I->Tok->isNot(tok::l_paren))
948 return false;
949 ++I;
950 if (I->Tok->isNot(Keywords.kw_function))
951 return false;
952 ++I;
953 return I->Tok->is(tok::l_paren);
954}
955
956static bool ShouldBreakBeforeBrace(const FormatStyle &Style,
957 const FormatToken &InitialToken,
958 bool IsEmptyBlock,
959 bool IsJavaRecord = false) {
960 if (IsJavaRecord)
961 return Style.BraceWrapping.AfterClass;
962
963 tok::TokenKind Kind = InitialToken.Tok.getKind();
964 if (InitialToken.is(TT_NamespaceMacro))
965 Kind = tok::kw_namespace;
966
967 const bool WrapRecordAllowed =
968 !IsEmptyBlock ||
969 Style.AllowShortRecordOnASingleLine < FormatStyle::SRS_Empty ||
970 Style.BraceWrapping.SplitEmptyRecord;
971
972 switch (Kind) {
973 case tok::kw_namespace:
974 return Style.BraceWrapping.AfterNamespace;
975 case tok::kw_class:
976 return Style.BraceWrapping.AfterClass && WrapRecordAllowed;
977 case tok::kw_union:
978 return Style.BraceWrapping.AfterUnion && WrapRecordAllowed;
979 case tok::kw_struct:
980 return Style.BraceWrapping.AfterStruct && WrapRecordAllowed;
981 case tok::kw_enum:
982 return Style.BraceWrapping.AfterEnum;
983 default:
984 return false;
985 }
986}
987
988void UnwrappedLineParser::parseChildBlock() {
989 assert(FormatTok->is(tok::l_brace));
990 FormatTok->setBlockKind(BK_Block);
991 const FormatToken *OpeningBrace = FormatTok;
992 nextToken();
993 {
994 bool SkipIndent = (Style.isJavaScript() &&
995 (isGoogScope(*Line) || isIIFE(*Line, Keywords)));
996 ScopedLineState LineState(*this);
997 ScopedDeclarationState DeclarationState(*Line, DeclarationScopeStack,
998 /*MustBeDeclaration=*/false);
999 Line->Level += SkipIndent ? 0 : 1;
1000 parseLevel(OpeningBrace);
1001 flushComments(isOnNewLine(*FormatTok));
1002 Line->Level -= SkipIndent ? 0 : 1;
1003 }
1004 nextToken();
1005}
1006
1007void UnwrappedLineParser::parsePPDirective() {
1008 assert(FormatTok->is(tok::hash) && "'#' expected");
1009 ScopedMacroState MacroState(*Line, Tokens, FormatTok);
1010
1011 nextToken();
1012
1013 if (!FormatTok->Tok.getIdentifierInfo()) {
1014 parsePPUnknown();
1015 return;
1016 }
1017
1018 switch (FormatTok->Tok.getIdentifierInfo()->getPPKeywordID()) {
1019 case tok::pp_define:
1020 parsePPDefine();
1021 return;
1022 case tok::pp_if:
1023 parsePPIf(/*IfDef=*/false);
1024 break;
1025 case tok::pp_ifdef:
1026 case tok::pp_ifndef:
1027 parsePPIf(/*IfDef=*/true);
1028 break;
1029 case tok::pp_else:
1030 case tok::pp_elifdef:
1031 case tok::pp_elifndef:
1032 case tok::pp_elif:
1033 parsePPElse();
1034 break;
1035 case tok::pp_endif:
1036 parsePPEndIf();
1037 break;
1038 case tok::pp_pragma:
1039 parsePPPragma();
1040 break;
1041 case tok::pp_error:
1042 case tok::pp_warning:
1043 nextToken();
1044 if (!eof() && Style.isCpp())
1045 FormatTok->setFinalizedType(TT_AfterPPDirective);
1046 [[fallthrough]];
1047 default:
1048 parsePPUnknown();
1049 break;
1050 }
1051}
1052
1053void UnwrappedLineParser::conditionalCompilationCondition(bool Unreachable) {
1054 size_t Line = CurrentLines->size();
1055 if (CurrentLines == &PreprocessorDirectives)
1056 Line += Lines.size();
1057
1058 if (Unreachable ||
1059 (!PPStack.empty() && PPStack.back().Kind == PP_Unreachable)) {
1060 PPStack.push_back({PP_Unreachable, Line});
1061 } else {
1062 PPStack.push_back({PP_Conditional, Line});
1063 }
1064}
1065
1066void UnwrappedLineParser::conditionalCompilationStart(bool Unreachable) {
1067 ++PPBranchLevel;
1068 assert(PPBranchLevel >= 0 && PPBranchLevel <= (int)PPLevelBranchIndex.size());
1069 if (PPBranchLevel == (int)PPLevelBranchIndex.size()) {
1070 PPLevelBranchIndex.push_back(0);
1071 PPLevelBranchCount.push_back(0);
1072 }
1073 PPChainBranchIndex.push(Unreachable ? -1 : 0);
1074 bool Skip = PPLevelBranchIndex[PPBranchLevel] > 0;
1075 conditionalCompilationCondition(Unreachable || Skip);
1076}
1077
1078void UnwrappedLineParser::conditionalCompilationAlternative() {
1079 if (!PPStack.empty())
1080 PPStack.pop_back();
1081 assert(PPBranchLevel < (int)PPLevelBranchIndex.size());
1082 if (!PPChainBranchIndex.empty())
1083 ++PPChainBranchIndex.top();
1084 conditionalCompilationCondition(
1085 PPBranchLevel >= 0 && !PPChainBranchIndex.empty() &&
1086 PPLevelBranchIndex[PPBranchLevel] != PPChainBranchIndex.top());
1087}
1088
1089void UnwrappedLineParser::conditionalCompilationEnd() {
1090 assert(PPBranchLevel < (int)PPLevelBranchIndex.size());
1091 if (PPBranchLevel >= 0 && !PPChainBranchIndex.empty()) {
1092 if (PPChainBranchIndex.top() + 1 > PPLevelBranchCount[PPBranchLevel])
1093 PPLevelBranchCount[PPBranchLevel] = PPChainBranchIndex.top() + 1;
1094 }
1095 // Guard against #endif's without #if.
1096 if (PPBranchLevel > -1)
1097 --PPBranchLevel;
1098 if (!PPChainBranchIndex.empty())
1099 PPChainBranchIndex.pop();
1100 if (!PPStack.empty())
1101 PPStack.pop_back();
1102}
1103
1104void UnwrappedLineParser::parsePPIf(bool IfDef) {
1105 bool IfNDef = FormatTok->is(tok::pp_ifndef);
1106 nextToken();
1107 bool Unreachable = false;
1108 if (!IfDef && (FormatTok->is(tok::kw_false) || FormatTok->TokenText == "0"))
1109 Unreachable = true;
1110 if (IfDef && !IfNDef && FormatTok->TokenText == "SWIG")
1111 Unreachable = true;
1112 conditionalCompilationStart(Unreachable);
1113 FormatToken *IfCondition = FormatTok;
1114 // If there's a #ifndef on the first line, and the only lines before it are
1115 // comments, it could be an include guard.
1116 bool MaybeIncludeGuard = IfNDef;
1117 if (IncludeGuard == IG_Inited && MaybeIncludeGuard) {
1118 for (auto &Line : Lines) {
1119 if (Line.Tokens.front().Tok->isNot(tok::comment)) {
1120 MaybeIncludeGuard = false;
1121 IncludeGuard = IG_Rejected;
1122 break;
1123 }
1124 }
1125 }
1126 --PPBranchLevel;
1127 parsePPUnknown();
1128 ++PPBranchLevel;
1129 if (IncludeGuard == IG_Inited && MaybeIncludeGuard) {
1130 IncludeGuard = IG_IfNdefed;
1131 IncludeGuardToken = IfCondition;
1132 }
1133}
1134
1135void UnwrappedLineParser::parsePPElse() {
1136 // If a potential include guard has an #else, it's not an include guard.
1137 if (IncludeGuard == IG_Defined && PPBranchLevel == 0)
1138 IncludeGuard = IG_Rejected;
1139 // Don't crash when there is an #else without an #if.
1140 assert(PPBranchLevel >= -1);
1141 if (PPBranchLevel == -1)
1142 conditionalCompilationStart(/*Unreachable=*/true);
1143 conditionalCompilationAlternative();
1144 --PPBranchLevel;
1145 parsePPUnknown();
1146 ++PPBranchLevel;
1147}
1148
1149void UnwrappedLineParser::parsePPEndIf() {
1150 conditionalCompilationEnd();
1151 parsePPUnknown();
1152}
1153
1154void UnwrappedLineParser::parsePPDefine() {
1155 nextToken();
1156
1157 if (!FormatTok->Tok.getIdentifierInfo()) {
1158 IncludeGuard = IG_Rejected;
1159 IncludeGuardToken = nullptr;
1160 parsePPUnknown();
1161 return;
1162 }
1163
1164 bool MaybeIncludeGuard = false;
1165 if (IncludeGuard == IG_IfNdefed &&
1166 IncludeGuardToken->TokenText == FormatTok->TokenText) {
1167 IncludeGuard = IG_Defined;
1168 IncludeGuardToken = nullptr;
1169 for (auto &Line : Lines) {
1170 if (Line.Tokens.front().Tok->isNoneOf(tok::comment, tok::hash)) {
1171 IncludeGuard = IG_Rejected;
1172 break;
1173 }
1174 }
1175 MaybeIncludeGuard = IncludeGuard == IG_Defined;
1176 }
1177
1178 // In the context of a define, even keywords should be treated as normal
1179 // identifiers. Setting the kind to identifier is not enough, because we need
1180 // to treat additional keywords like __except as well, which are already
1181 // identifiers. Setting the identifier info to null interferes with include
1182 // guard processing above, and changes preprocessing nesting.
1183 FormatTok->Tok.setKind(tok::identifier);
1184 FormatTok->Tok.setIdentifierInfo(Keywords.kw_internal_ident_after_define);
1185 nextToken();
1186
1187 // IncludeGuard can't have a non-empty macro definition.
1188 if (MaybeIncludeGuard && !eof())
1189 IncludeGuard = IG_Rejected;
1190
1191 if (FormatTok->is(tok::l_paren) && !FormatTok->hasWhitespaceBefore())
1192 parseParens();
1193 if (Style.IndentPPDirectives != FormatStyle::PPDIS_None)
1194 Line->Level += PPBranchLevel + 1;
1195 addUnwrappedLine();
1196 ++Line->Level;
1197
1198 Line->PPLevel = PPBranchLevel + (IncludeGuard == IG_Defined ? 0 : 1);
1199 assert((int)Line->PPLevel >= 0);
1200
1201 if (eof())
1202 return;
1203
1204 Line->InMacroBody = true;
1205
1206 if (!Style.SkipMacroDefinitionBody) {
1207 // Errors during a preprocessor directive can only affect the layout of the
1208 // preprocessor directive, and thus we ignore them. An alternative approach
1209 // would be to use the same approach we use on the file level (no
1210 // re-indentation if there was a structural error) within the macro
1211 // definition.
1212 parseFile();
1213 return;
1214 }
1215
1216 for (auto *Comment : CommentsBeforeNextToken)
1217 Comment->Finalized = true;
1218
1219 do {
1220 FormatTok->Finalized = true;
1221 FormatTok = Tokens->getNextToken();
1222 } while (!eof());
1223
1224 addUnwrappedLine();
1225}
1226
1227void UnwrappedLineParser::parsePPPragma() {
1228 Line->InPragmaDirective = true;
1229 parsePPUnknown();
1230}
1231
1232void UnwrappedLineParser::parsePPUnknown() {
1233 while (!eof())
1234 nextToken();
1235 if (Style.IndentPPDirectives != FormatStyle::PPDIS_None)
1236 Line->Level += PPBranchLevel + 1;
1237 addUnwrappedLine();
1238}
1239
1240// Here we exclude certain tokens that are not usually the first token in an
1241// unwrapped line. This is used in attempt to distinguish macro calls without
1242// trailing semicolons from other constructs split to several lines.
1244 // Semicolon can be a null-statement, l_square can be a start of a macro or
1245 // a C++11 attribute, but this doesn't seem to be common.
1246 return Tok.isNoneOf(tok::semi, tok::l_brace,
1247 // Tokens that can only be used as binary operators and a
1248 // part of overloaded operator names.
1249 tok::period, tok::periodstar, tok::arrow, tok::arrowstar,
1250 tok::less, tok::greater, tok::slash, tok::percent,
1251 tok::lessless, tok::greatergreater, tok::equal,
1252 tok::plusequal, tok::minusequal, tok::starequal,
1253 tok::slashequal, tok::percentequal, tok::ampequal,
1254 tok::pipeequal, tok::caretequal, tok::greatergreaterequal,
1255 tok::lesslessequal,
1256 // Colon is used in labels, base class lists, initializer
1257 // lists, range-based for loops, ternary operator, but
1258 // should never be the first token in an unwrapped line.
1259 tok::colon,
1260 // 'noexcept' is a trailing annotation.
1261 tok::kw_noexcept);
1262}
1263
1264static bool mustBeJSIdent(const AdditionalKeywords &Keywords,
1265 const FormatToken *FormatTok) {
1266 // FIXME: This returns true for C/C++ keywords like 'struct'.
1267 return FormatTok->is(tok::identifier) &&
1268 (!FormatTok->Tok.getIdentifierInfo() ||
1269 FormatTok->isNoneOf(
1270 Keywords.kw_in, Keywords.kw_of, Keywords.kw_as, Keywords.kw_async,
1271 Keywords.kw_await, Keywords.kw_yield, Keywords.kw_finally,
1272 Keywords.kw_function, Keywords.kw_import, Keywords.kw_is,
1273 Keywords.kw_let, Keywords.kw_var, tok::kw_const,
1274 Keywords.kw_abstract, Keywords.kw_extends, Keywords.kw_implements,
1275 Keywords.kw_instanceof, Keywords.kw_interface,
1276 Keywords.kw_override, Keywords.kw_throws, Keywords.kw_from));
1277}
1278
1279static bool mustBeJSIdentOrValue(const AdditionalKeywords &Keywords,
1280 const FormatToken *FormatTok) {
1281 return FormatTok->Tok.isLiteral() ||
1282 FormatTok->isOneOf(tok::kw_true, tok::kw_false) ||
1283 mustBeJSIdent(Keywords, FormatTok);
1284}
1285
1286// isJSDeclOrStmt returns true if |FormatTok| starts a declaration or statement
1287// when encountered after a value (see mustBeJSIdentOrValue).
1288static bool isJSDeclOrStmt(const AdditionalKeywords &Keywords,
1289 const FormatToken *FormatTok) {
1290 return FormatTok->isOneOf(
1291 tok::kw_return, Keywords.kw_yield,
1292 // conditionals
1293 tok::kw_if, tok::kw_else,
1294 // loops
1295 tok::kw_for, tok::kw_while, tok::kw_do, tok::kw_continue, tok::kw_break,
1296 // switch/case
1297 tok::kw_switch, tok::kw_case,
1298 // exceptions
1299 tok::kw_throw, tok::kw_try, tok::kw_catch, Keywords.kw_finally,
1300 // declaration
1301 tok::kw_const, tok::kw_class, Keywords.kw_var, Keywords.kw_let,
1302 Keywords.kw_async, Keywords.kw_function,
1303 // import/export
1304 Keywords.kw_import, tok::kw_export);
1305}
1306
1307// Checks whether a token is a type in K&R C (aka C78).
1308static bool isC78Type(const FormatToken &Tok) {
1309 return Tok.isOneOf(tok::kw_char, tok::kw_short, tok::kw_int, tok::kw_long,
1310 tok::kw_unsigned, tok::kw_float, tok::kw_double,
1311 tok::identifier);
1312}
1313
1314// This function checks whether a token starts the first parameter declaration
1315// in a K&R C (aka C78) function definition, e.g.:
1316// int f(a, b)
1317// short a, b;
1318// {
1319// return a + b;
1320// }
1322 const FormatToken *FuncName) {
1323 assert(Tok);
1324 assert(Next);
1325 assert(FuncName);
1326
1327 if (FuncName->isNot(tok::identifier))
1328 return false;
1329
1330 const FormatToken *Prev = FuncName->Previous;
1331 if (!Prev || (Prev->isNot(tok::star) && !isC78Type(*Prev)))
1332 return false;
1333
1334 if (!isC78Type(*Tok) &&
1335 Tok->isNoneOf(tok::kw_register, tok::kw_struct, tok::kw_union)) {
1336 return false;
1337 }
1338
1339 if (Next->isNot(tok::star) && !Next->Tok.getIdentifierInfo())
1340 return false;
1341
1342 Tok = Tok->Previous;
1343 if (!Tok || Tok->isNot(tok::r_paren))
1344 return false;
1345
1346 Tok = Tok->Previous;
1347 if (!Tok || Tok->isNot(tok::identifier))
1348 return false;
1349
1350 return Tok->Previous && Tok->Previous->isOneOf(tok::l_paren, tok::comma);
1351}
1352
1353bool UnwrappedLineParser::parseModuleDecl() {
1354 assert(IsCpp);
1355 assert(FormatTok->is(Keywords.kw_module));
1356
1357 if (Style.Language == FormatStyle::LK_C ||
1358 Style.Standard < FormatStyle::LS_Cpp20) {
1359 return false;
1360 }
1361
1362 nextToken();
1363 if (FormatTok->isNot(tok::identifier))
1364 return false;
1365
1366 for (nextToken(); FormatTok->isNoneOf(tok::semi, tok::eof); nextToken())
1367 if (FormatTok->is(tok::colon))
1368 FormatTok->setFinalizedType(TT_ModulePartitionColon);
1369
1370 nextToken();
1371 Line->IsModuleOrImportDecl = true;
1372 addUnwrappedLine();
1373 return true;
1374}
1375
1376bool UnwrappedLineParser::parseImportDecl() {
1377 assert(IsCpp);
1378 assert(FormatTok->is(Keywords.kw_import) && "'import' expected");
1379
1380 if (Style.Language == FormatStyle::LK_C ||
1381 Style.Standard < FormatStyle::LS_Cpp20) {
1382 return false;
1383 }
1384
1385 nextToken();
1386 if (FormatTok->is(tok::colon)) {
1387 FormatTok->setFinalizedType(TT_ModulePartitionColon);
1388 nextToken();
1389 }
1390 if (FormatTok->isNoneOf(tok::identifier, tok::less, tok::string_literal))
1391 return false;
1392
1393 for (; FormatTok->isNoneOf(tok::semi, tok::eof); nextToken()) {
1394 // Handle import <foo/bar.h> as we would an include statement.
1395 if (FormatTok->is(tok::less)) {
1396 for (nextToken(); FormatTok->isNoneOf(tok::greater, tok::semi, tok::eof);
1397 nextToken()) {
1398 // Mark tokens as implicit string literals, so that import <A/Foo> will
1399 // neither be broken nor have a space added.
1400 FormatTok->setFinalizedType(TT_ImplicitStringLiteral);
1401 }
1402 }
1403 }
1404
1405 nextToken();
1406 Line->IsModuleOrImportDecl = true;
1407 addUnwrappedLine();
1408 return true;
1409}
1410
1411// readTokenWithJavaScriptASI reads the next token and terminates the current
1412// line if JavaScript Automatic Semicolon Insertion must
1413// happen between the current token and the next token.
1414//
1415// This method is conservative - it cannot cover all edge cases of JavaScript,
1416// but only aims to correctly handle certain well known cases. It *must not*
1417// return true in speculative cases.
1418void UnwrappedLineParser::readTokenWithJavaScriptASI() {
1419 FormatToken *Previous = FormatTok;
1420 readToken();
1421 FormatToken *Next = FormatTok;
1422
1423 bool IsOnSameLine =
1424 CommentsBeforeNextToken.empty()
1425 ? Next->NewlinesBefore == 0
1426 : CommentsBeforeNextToken.front()->NewlinesBefore == 0;
1427 if (IsOnSameLine)
1428 return;
1429
1430 bool PreviousMustBeValue = mustBeJSIdentOrValue(Keywords, Previous);
1431 bool PreviousStartsTemplateExpr =
1432 Previous->is(TT_TemplateString) && Previous->TokenText.ends_with("${");
1433 if (PreviousMustBeValue || Previous->is(tok::r_paren)) {
1434 // If the line contains an '@' sign, the previous token might be an
1435 // annotation, which can precede another identifier/value.
1436 bool HasAt = llvm::any_of(Line->Tokens, [](UnwrappedLineNode &LineNode) {
1437 return LineNode.Tok->is(tok::at);
1438 });
1439 if (HasAt)
1440 return;
1441 }
1442 if (Next->is(tok::exclaim) && PreviousMustBeValue)
1443 return addUnwrappedLine();
1444 bool NextMustBeValue = mustBeJSIdentOrValue(Keywords, Next);
1445 bool NextEndsTemplateExpr =
1446 Next->is(TT_TemplateString) && Next->TokenText.starts_with("}");
1447 if (NextMustBeValue && !NextEndsTemplateExpr && !PreviousStartsTemplateExpr &&
1448 (PreviousMustBeValue ||
1449 Previous->isOneOf(tok::r_square, tok::r_paren, tok::plusplus,
1450 tok::minusminus))) {
1451 return addUnwrappedLine();
1452 }
1453 if ((PreviousMustBeValue || Previous->is(tok::r_paren)) &&
1454 isJSDeclOrStmt(Keywords, Next)) {
1455 return addUnwrappedLine();
1456 }
1457}
1458
1459void UnwrappedLineParser::parseStructuralElement(
1460 const FormatToken *OpeningBrace, IfStmtKind *IfKind,
1461 FormatToken **IfLeftBrace, bool *HasDoWhile, bool *HasLabel) {
1462 if (Style.isTableGen() && FormatTok->is(tok::pp_include)) {
1463 nextToken();
1464 if (FormatTok->is(tok::string_literal))
1465 nextToken();
1466 addUnwrappedLine();
1467 return;
1468 }
1469
1470 if (IsCpp) {
1471 while (FormatTok->is(tok::l_square) && handleCppAttributes()) {
1472 }
1473 } else if (Style.isVerilog()) {
1474 // Skip attributes.
1475 while (FormatTok->is(tok::l_paren) &&
1476 Tokens->peekNextToken()->is(tok::star)) {
1477 parseParens();
1478 }
1479 skipVerilogQualifiers();
1480 // Skip things that can exist before keywords like 'if' and 'case'.
1481 if (FormatTok->isOneOf(Keywords.kw_priority, Keywords.kw_unique,
1482 Keywords.kw_unique0)) {
1483 nextToken();
1484 }
1485
1486 if (Keywords.isVerilogStructuredProcedure(*FormatTok)) {
1487 parseForOrWhileLoop(/*HasParens=*/false);
1488 return;
1489 }
1490 if (FormatTok->isOneOf(Keywords.kw_foreach, Keywords.kw_repeat)) {
1491 parseForOrWhileLoop();
1492 return;
1493 }
1494 if (FormatTok->isOneOf(tok::kw_restrict, Keywords.kw_assert,
1495 Keywords.kw_assume, Keywords.kw_cover)) {
1496 parseIfThenElse(IfKind, /*KeepBraces=*/false, /*IsVerilogAssert=*/true);
1497 return;
1498 }
1499 }
1500
1501 // Tokens that only make sense at the beginning of a line.
1502 if (FormatTok->isAccessSpecifierKeyword()) {
1503 if (Style.isJava() || Style.isJavaScript() || Style.isCSharp())
1504 nextToken();
1505 else
1506 parseAccessSpecifier();
1507 return;
1508 }
1509 switch (FormatTok->Tok.getKind()) {
1510 case tok::kw_asm: {
1511 // Track whether to skip formatting inline asm by finalizing the tokens
1512 // in the block. Formatting is skipped inside of braces by default.
1513 // A style option could be added to also skip formatting inside parens.
1514 bool DoNotFormat = false;
1515 tok::TokenKind OpenType;
1516 tok::TokenKind CloseType;
1517 nextToken();
1518 while (FormatTok &&
1519 FormatTok->isOneOf(tok::kw_volatile, tok::kw_inline, tok::kw_goto)) {
1520 nextToken();
1521 }
1522 if (!FormatTok)
1523 break;
1524 if (FormatTok->is(tok::l_brace)) {
1525 FormatTok->setFinalizedType(TT_InlineASMBrace);
1526 OpenType = tok::l_brace;
1527 CloseType = tok::r_brace;
1528 DoNotFormat = true;
1529 } else if (FormatTok->is(tok::l_paren)) {
1530 OpenType = tok::l_paren;
1531 CloseType = tok::r_paren;
1532 FormatTok->setFinalizedType(TT_InlineASMParen);
1533 } else {
1534 break;
1535 }
1536 if (DoNotFormat) {
1537 FormatToken *OpenTok = FormatTok;
1538 int NestLevel = 0;
1539 nextToken();
1540 while (FormatTok && !eof()) {
1541 if (FormatTok->is(OpenType)) {
1542 ++NestLevel;
1543 } else if (FormatTok->is(CloseType)) {
1544 --NestLevel;
1545 if (NestLevel < 1) {
1546 FormatTok->setFinalizedType(OpenTok->getType());
1547 nextToken();
1548 addUnwrappedLine();
1549 break;
1550 }
1551 }
1552 FormatTok->Finalized = true;
1553 nextToken();
1554 }
1555 }
1556 break;
1557 }
1558 case tok::kw_namespace:
1559 parseNamespace();
1560 return;
1561 case tok::kw_if: {
1562 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1563 // field/method declaration.
1564 break;
1565 }
1566 FormatToken *Tok = parseIfThenElse(IfKind);
1567 if (IfLeftBrace)
1568 *IfLeftBrace = Tok;
1569 return;
1570 }
1571 case tok::kw_for:
1572 case tok::kw_while:
1573 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1574 // field/method declaration.
1575 break;
1576 }
1577 parseForOrWhileLoop();
1578 return;
1579 case tok::kw_do:
1580 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1581 // field/method declaration.
1582 break;
1583 }
1584 parseDoWhile();
1585 if (HasDoWhile)
1586 *HasDoWhile = true;
1587 return;
1588 case tok::kw_switch:
1589 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1590 // 'switch: string' field declaration.
1591 break;
1592 }
1593 parseSwitch(/*IsExpr=*/false);
1594 return;
1595 case tok::kw_default: {
1596 // In Verilog default along with other labels are handled in the next loop.
1597 if (Style.isVerilog())
1598 break;
1599 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1600 // 'default: string' field declaration.
1601 break;
1602 }
1603 auto *Default = FormatTok;
1604 nextToken();
1605 if (FormatTok->is(tok::colon)) {
1606 FormatTok->setFinalizedType(TT_CaseLabelColon);
1607 parseLabel();
1608 return;
1609 }
1610 if (FormatTok->is(tok::arrow)) {
1611 FormatTok->setFinalizedType(TT_CaseLabelArrow);
1612 Default->setFinalizedType(TT_SwitchExpressionLabel);
1613 parseLabel();
1614 return;
1615 }
1616 // e.g. "default void f() {}" in a Java interface.
1617 break;
1618 }
1619 case tok::kw_case:
1620 // Proto: there are no switch/case statements.
1621 if (Style.Language == FormatStyle::LK_Proto) {
1622 nextToken();
1623 return;
1624 }
1625 if (Style.isVerilog()) {
1626 parseBlock();
1627 addUnwrappedLine();
1628 return;
1629 }
1630 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1631 // 'case: string' field declaration.
1632 nextToken();
1633 break;
1634 }
1635 parseCaseLabel();
1636 return;
1637 case tok::kw_goto:
1638 nextToken();
1639 if (FormatTok->is(tok::kw_case))
1640 nextToken();
1641 break;
1642 case tok::kw_try:
1643 case tok::kw___try:
1644 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1645 // field/method declaration.
1646 break;
1647 }
1648 parseTryCatch();
1649 return;
1650 case tok::kw_extern:
1651 if (Style.isVerilog()) {
1652 // In Verilog an extern module declaration looks like a start of module.
1653 // But there is no body and endmodule. So we handle it separately.
1654 parseVerilogExtern();
1655 return;
1656 }
1657 nextToken();
1658 if (FormatTok->is(tok::string_literal)) {
1659 nextToken();
1660 if (FormatTok->is(tok::l_brace)) {
1661 if (Style.BraceWrapping.AfterExternBlock)
1662 addUnwrappedLine();
1663 // Either we indent or for backwards compatibility we follow the
1664 // AfterExternBlock style.
1665 unsigned AddLevels =
1666 (Style.IndentExternBlock == FormatStyle::IEBS_Indent) ||
1667 (Style.BraceWrapping.AfterExternBlock &&
1668 Style.IndentExternBlock ==
1670 ? 1u
1671 : 0u;
1672 parseBlock(/*MustBeDeclaration=*/true, AddLevels);
1673 addUnwrappedLine();
1674 return;
1675 }
1676 }
1677 break;
1678 case tok::kw_export:
1679 if (IsCpp) {
1680 nextToken();
1681 if (FormatTok->is(tok::kw_namespace)) {
1682 parseNamespace();
1683 return;
1684 }
1685 if (FormatTok->is(tok::l_brace)) {
1686 parseCppExportBlock();
1687 return;
1688 }
1689 if (FormatTok->is(Keywords.kw_module) && parseModuleDecl())
1690 return;
1691 if (FormatTok->is(Keywords.kw_import) && parseImportDecl())
1692 return;
1693 break;
1694 }
1695 if (Style.isJavaScript()) {
1696 parseJavaScriptEs6ImportExport();
1697 return;
1698 }
1699 if (Style.isVerilog()) {
1700 parseVerilogExtern();
1701 return;
1702 }
1703 break;
1704 case tok::kw_inline:
1705 nextToken();
1706 if (FormatTok->is(tok::kw_namespace)) {
1707 parseNamespace();
1708 return;
1709 }
1710 break;
1711 case tok::identifier:
1712 if (FormatTok->is(TT_ForEachMacro)) {
1713 parseForOrWhileLoop();
1714 return;
1715 }
1716 if (FormatTok->is(TT_MacroBlockBegin)) {
1717 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
1718 /*MunchSemi=*/false);
1719 return;
1720 }
1721 if (FormatTok->is(Keywords.kw_import)) {
1722 if (IsCpp && parseImportDecl())
1723 return;
1724 if (Style.isJavaScript()) {
1725 parseJavaScriptEs6ImportExport();
1726 return;
1727 }
1728 if (Style.Language == FormatStyle::LK_Proto) {
1729 nextToken();
1730 if (FormatTok->is(tok::kw_public))
1731 nextToken();
1732 if (FormatTok->isNot(tok::string_literal))
1733 return;
1734 nextToken();
1735 if (FormatTok->is(tok::semi))
1736 nextToken();
1737 addUnwrappedLine();
1738 return;
1739 }
1740 if (Style.isVerilog()) {
1741 parseVerilogExtern();
1742 return;
1743 }
1744 }
1745 if (IsCpp) {
1746 if (FormatTok->is(Keywords.kw_module) && parseModuleDecl())
1747 return;
1748 if (FormatTok->isOneOf(Keywords.kw_signals, Keywords.kw_qsignals,
1749 Keywords.kw_slots, Keywords.kw_qslots)) {
1750 nextToken();
1751 if (FormatTok->is(tok::colon)) {
1752 nextToken();
1753 addUnwrappedLine();
1754 return;
1755 }
1756 }
1757 if (FormatTok->is(TT_StatementMacro)) {
1758 parseStatementMacro();
1759 return;
1760 }
1761 if (FormatTok->is(TT_NamespaceMacro)) {
1762 parseNamespace();
1763 return;
1764 }
1765 }
1766 // In Verilog labels can be any expression, so we don't do them here.
1767 // JS doesn't have macros, and within classes colons indicate fields, not
1768 // labels.
1769 // TableGen doesn't have labels.
1770 if (!Style.isJavaScript() && !Style.isVerilog() && !Style.isTableGen() &&
1771 Tokens->peekNextToken()->is(tok::colon) && !Line->MustBeDeclaration) {
1772 nextToken();
1773 if (!Line->InMacroBody || CurrentLines->size() > 1)
1774 Line->Tokens.begin()->Tok->MustBreakBefore = true;
1775 FormatTok->setFinalizedType(TT_GotoLabelColon);
1776 parseLabel(/*IsGotoLabel=*/true);
1777 if (HasLabel)
1778 *HasLabel = true;
1779 return;
1780 }
1781 if (Style.isJava() && FormatTok->is(Keywords.kw_record)) {
1782 parseRecord(/*ParseAsExpr=*/false, /*IsJavaRecord=*/true);
1783 addUnwrappedLine();
1784 return;
1785 }
1786 // In all other cases, parse the declaration.
1787 break;
1788 default:
1789 break;
1790 }
1791
1792 bool SeenEqual = false;
1793 for (const bool InRequiresExpression =
1794 OpeningBrace && OpeningBrace->isOneOf(TT_RequiresExpressionLBrace,
1795 TT_CompoundRequirementLBrace);
1796 !eof();) {
1797 const FormatToken *Previous = FormatTok->Previous;
1798 switch (FormatTok->Tok.getKind()) {
1799 case tok::at:
1800 nextToken();
1801 if (FormatTok->is(tok::l_brace)) {
1802 nextToken();
1803 parseBracedList();
1804 break;
1805 }
1806 if (Style.isJava() && FormatTok->is(Keywords.kw_interface)) {
1807 nextToken();
1808 break;
1809 }
1810 switch (bool IsAutoRelease = false; FormatTok->Tok.getObjCKeywordID()) {
1811 case tok::objc_public:
1812 case tok::objc_protected:
1813 case tok::objc_package:
1814 case tok::objc_private:
1815 return parseAccessSpecifier();
1816 case tok::objc_interface:
1817 case tok::objc_implementation:
1818 return parseObjCInterfaceOrImplementation();
1819 case tok::objc_protocol:
1820 if (parseObjCProtocol())
1821 return;
1822 break;
1823 case tok::objc_end:
1824 return; // Handled by the caller.
1825 case tok::objc_optional:
1826 case tok::objc_required:
1827 nextToken();
1828 addUnwrappedLine();
1829 return;
1830 case tok::objc_autoreleasepool:
1831 IsAutoRelease = true;
1832 [[fallthrough]];
1833 case tok::objc_synchronized:
1834 nextToken();
1835 if (!IsAutoRelease && FormatTok->is(tok::l_paren)) {
1836 // Skip synchronization object
1837 parseParens();
1838 }
1839 if (FormatTok->is(tok::l_brace)) {
1840 if (Style.BraceWrapping.AfterControlStatement ==
1842 addUnwrappedLine();
1843 }
1844 parseBlock();
1845 }
1846 addUnwrappedLine();
1847 return;
1848 case tok::objc_try:
1849 // This branch isn't strictly necessary (the kw_try case below would
1850 // do this too after the tok::at is parsed above). But be explicit.
1851 parseTryCatch();
1852 return;
1853 default:
1854 break;
1855 }
1856 break;
1857 case tok::kw_requires: {
1858 if (IsCpp) {
1859 bool ParsedClause = parseRequires(SeenEqual);
1860 if (ParsedClause)
1861 return;
1862 } else {
1863 nextToken();
1864 }
1865 break;
1866 }
1867 case tok::kw_enum:
1868 // Ignore if this is part of "template <enum ..." or "... -> enum" or
1869 // "template <..., enum ...>".
1870 if (Previous && Previous->isOneOf(tok::less, tok::arrow, tok::comma)) {
1871 nextToken();
1872 break;
1873 }
1874
1875 // parseEnum falls through and does not yet add an unwrapped line as an
1876 // enum definition can start a structural element.
1877 if (!parseEnum())
1878 break;
1879 // This only applies to C++ and Verilog.
1880 if (!IsCpp && !Style.isVerilog()) {
1881 addUnwrappedLine();
1882 return;
1883 }
1884 break;
1885 case tok::kw_typedef:
1886 nextToken();
1887 if (FormatTok->isOneOf(Keywords.kw_NS_ENUM, Keywords.kw_NS_OPTIONS,
1888 Keywords.kw_CF_ENUM, Keywords.kw_CF_OPTIONS,
1889 Keywords.kw_CF_CLOSED_ENUM,
1890 Keywords.kw_NS_CLOSED_ENUM)) {
1891 parseEnum();
1892 }
1893 break;
1894 case tok::kw_class:
1895 if (Style.isVerilog()) {
1896 parseBlock();
1897 addUnwrappedLine();
1898 return;
1899 }
1900 if (Style.isTableGen()) {
1901 // Do nothing special. In this case the l_brace becomes FunctionLBrace.
1902 // This is same as def and so on.
1903 nextToken();
1904 break;
1905 }
1906 [[fallthrough]];
1907 case tok::kw_struct:
1908 case tok::kw_union:
1909 if (parseStructLike())
1910 return;
1911 break;
1912 case tok::kw_decltype:
1913 nextToken();
1914 if (FormatTok->is(tok::l_paren)) {
1915 parseParens();
1916 if (FormatTok->Previous &&
1917 FormatTok->Previous->endsSequence(tok::r_paren, tok::kw_auto,
1918 tok::l_paren)) {
1919 Line->SeenDecltypeAuto = true;
1920 }
1921 }
1922 break;
1923 case tok::period:
1924 nextToken();
1925 // In Java, classes have an implicit static member "class".
1926 if (Style.isJava() && FormatTok && FormatTok->is(tok::kw_class))
1927 nextToken();
1928 if (Style.isJavaScript() && FormatTok &&
1929 FormatTok->Tok.getIdentifierInfo()) {
1930 // JavaScript only has pseudo keywords, all keywords are allowed to
1931 // appear in "IdentifierName" positions. See http://es5.github.io/#x7.6
1932 nextToken();
1933 }
1934 break;
1935 case tok::semi:
1936 nextToken();
1937 addUnwrappedLine();
1938 return;
1939 case tok::r_brace:
1940 addUnwrappedLine();
1941 return;
1942 case tok::string_literal:
1943 if (Style.isVerilog() && FormatTok->is(TT_VerilogProtected)) {
1944 FormatTok->Finalized = true;
1945 nextToken();
1946 addUnwrappedLine();
1947 return;
1948 }
1949 nextToken();
1950 break;
1951 case tok::l_paren: {
1952 parseParens();
1953 // Break the unwrapped line if a K&R C function definition has a parameter
1954 // declaration.
1955 if (OpeningBrace || !IsCpp || !Previous || eof())
1956 break;
1957 if (isC78ParameterDecl(FormatTok,
1958 Tokens->peekNextToken(/*SkipComment=*/true),
1959 Previous)) {
1960 addUnwrappedLine();
1961 return;
1962 }
1963 break;
1964 }
1965 case tok::kw_operator:
1966 nextToken();
1967 if (FormatTok->isBinaryOperator())
1968 nextToken();
1969 break;
1970 case tok::caret: {
1971 const auto *Prev = FormatTok->getPreviousNonComment();
1972 nextToken();
1973 if (Prev && Prev->is(tok::identifier))
1974 break;
1975 // Block return type.
1976 if (FormatTok->Tok.isAnyIdentifier() || FormatTok->isTypeName(LangOpts)) {
1977 nextToken();
1978 // Return types: pointers are ok too.
1979 while (FormatTok->is(tok::star))
1980 nextToken();
1981 }
1982 // Block argument list.
1983 if (FormatTok->is(tok::l_paren))
1984 parseParens();
1985 // Block body.
1986 if (FormatTok->is(tok::l_brace))
1987 parseChildBlock();
1988 break;
1989 }
1990 case tok::l_brace:
1991 if (InRequiresExpression)
1992 FormatTok->setFinalizedType(TT_BracedListLBrace);
1993 if (!tryToParsePropertyAccessor() && !tryToParseBracedList()) {
1994 IsDecltypeAutoFunction = Line->SeenDecltypeAuto;
1995 // A block outside of parentheses must be the last part of a
1996 // structural element.
1997 // FIXME: Figure out cases where this is not true, and add projections
1998 // for them (the one we know is missing are lambdas).
1999 if (Style.isJava() &&
2000 Line->Tokens.front().Tok->is(Keywords.kw_synchronized)) {
2001 // If necessary, we could set the type to something different than
2002 // TT_FunctionLBrace.
2003 if (Style.BraceWrapping.AfterControlStatement ==
2005 addUnwrappedLine();
2006 }
2007 } else if (Style.BraceWrapping.AfterFunction) {
2008 addUnwrappedLine();
2009 }
2010 if (!Previous || Previous->isNot(TT_TypeDeclarationParen))
2011 FormatTok->setFinalizedType(TT_FunctionLBrace);
2012 parseBlock();
2013 IsDecltypeAutoFunction = false;
2014 addUnwrappedLine();
2015 return;
2016 }
2017 // Otherwise this was a braced init list, and the structural
2018 // element continues.
2019 break;
2020 case tok::kw_try:
2021 if (Style.isJavaScript() && Line->MustBeDeclaration) {
2022 // field/method declaration.
2023 nextToken();
2024 break;
2025 }
2026 // We arrive here when parsing function-try blocks.
2027 if (Style.BraceWrapping.AfterFunction)
2028 addUnwrappedLine();
2029 parseTryCatch();
2030 return;
2031 case tok::identifier: {
2032 if (Style.isCSharp() && FormatTok->is(Keywords.kw_where) &&
2033 Line->MustBeDeclaration) {
2034 addUnwrappedLine();
2035 parseCSharpGenericTypeConstraint();
2036 break;
2037 }
2038 if (FormatTok->is(TT_MacroBlockEnd)) {
2039 addUnwrappedLine();
2040 return;
2041 }
2042
2043 // Function declarations (as opposed to function expressions) are parsed
2044 // on their own unwrapped line by continuing this loop. Function
2045 // expressions (functions that are not on their own line) must not create
2046 // a new unwrapped line, so they are special cased below.
2047 size_t TokenCount = Line->Tokens.size();
2048 if (Style.isJavaScript() && FormatTok->is(Keywords.kw_function) &&
2049 (TokenCount > 1 ||
2050 (TokenCount == 1 &&
2051 Line->Tokens.front().Tok->isNot(Keywords.kw_async)))) {
2052 tryToParseJSFunction();
2053 break;
2054 }
2055 if ((Style.isJavaScript() || Style.isJava()) &&
2056 FormatTok->is(Keywords.kw_interface)) {
2057 if (Style.isJavaScript()) {
2058 // In JavaScript/TypeScript, "interface" can be used as a standalone
2059 // identifier, e.g. in `var interface = 1;`. If "interface" is
2060 // followed by another identifier, it is very like to be an actual
2061 // interface declaration.
2062 unsigned StoredPosition = Tokens->getPosition();
2063 FormatToken *Next = Tokens->getNextToken();
2064 FormatTok = Tokens->setPosition(StoredPosition);
2065 if (!mustBeJSIdent(Keywords, Next)) {
2066 nextToken();
2067 break;
2068 }
2069 }
2070 parseRecord();
2071 addUnwrappedLine();
2072 return;
2073 }
2074
2075 if (Style.isVerilog()) {
2076 if (FormatTok->is(Keywords.kw_table)) {
2077 parseVerilogTable();
2078 return;
2079 }
2080 if (Keywords.isVerilogBegin(*FormatTok) ||
2081 Keywords.isVerilogHierarchy(*FormatTok)) {
2082 parseBlock();
2083 addUnwrappedLine();
2084 return;
2085 }
2086 }
2087
2088 if (!IsCpp && FormatTok->is(Keywords.kw_interface)) {
2089 if (parseStructLike())
2090 return;
2091 break;
2092 }
2093
2094 if (IsCpp && FormatTok->is(TT_StatementMacro)) {
2095 parseStatementMacro();
2096 return;
2097 }
2098
2099 // See if the following token should start a new unwrapped line.
2100 StringRef Text = FormatTok->TokenText;
2101
2102 FormatToken *PreviousToken = FormatTok;
2103 nextToken();
2104
2105 // JS doesn't have macros, and within classes colons indicate fields, not
2106 // labels.
2107 if (Style.isJavaScript())
2108 break;
2109
2110 auto OneTokenSoFar = [&]() {
2111 auto I = Line->Tokens.begin(), E = Line->Tokens.end();
2112 while (I != E && I->Tok->is(tok::comment))
2113 ++I;
2114 if (Style.isVerilog())
2115 while (I != E && I->Tok->is(tok::hash))
2116 ++I;
2117 return I != E && (++I == E);
2118 };
2119 if (OneTokenSoFar()) {
2120 // Recognize function-like macro usages without trailing semicolon as
2121 // well as free-standing macros like Q_OBJECT.
2122 bool FunctionLike = FormatTok->is(tok::l_paren);
2123 if (FunctionLike)
2124 parseParens();
2125
2126 bool FollowedByNewline =
2127 CommentsBeforeNextToken.empty()
2128 ? FormatTok->NewlinesBefore > 0
2129 : CommentsBeforeNextToken.front()->NewlinesBefore > 0;
2130
2131 if (FollowedByNewline &&
2132 (Text.size() >= 5 ||
2133 (FunctionLike && FormatTok->isNot(tok::l_paren))) &&
2134 tokenCanStartNewLine(*FormatTok) && Text == Text.upper()) {
2135 if (PreviousToken->isNot(TT_UntouchableMacroFunc))
2136 PreviousToken->setFinalizedType(TT_FunctionLikeOrFreestandingMacro);
2137 addUnwrappedLine();
2138 return;
2139 }
2140 }
2141 break;
2142 }
2143 case tok::equal:
2144 if ((Style.isJavaScript() || Style.isCSharp()) &&
2145 FormatTok->is(TT_FatArrow)) {
2146 tryToParseChildBlock();
2147 break;
2148 }
2149
2150 SeenEqual = true;
2151 nextToken();
2152 if (FormatTok->is(tok::l_brace)) {
2153 // C# needs this change to ensure that array initialisers and object
2154 // initialisers are indented the same way. In TypeScript, the brace
2155 // can also be an object type definition.
2156 if (!Style.isJavaScript())
2157 FormatTok->setBlockKind(BK_BracedInit);
2158 // TableGen's defset statement has syntax of the form,
2159 // `defset <type> <name> = { <statement>... }`
2160 if (Style.isTableGen() &&
2161 Line->Tokens.begin()->Tok->is(Keywords.kw_defset)) {
2162 FormatTok->setFinalizedType(TT_FunctionLBrace);
2163 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
2164 /*MunchSemi=*/false);
2165 addUnwrappedLine();
2166 break;
2167 }
2168 nextToken();
2169 parseBracedList();
2170 } else if (Style.Language == FormatStyle::LK_Proto &&
2171 FormatTok->is(tok::less)) {
2172 nextToken();
2173 parseBracedList(/*IsAngleBracket=*/true);
2174 }
2175 break;
2176 case tok::l_square:
2177 parseSquare();
2178 break;
2179 case tok::kw_new:
2180 if (Style.isCSharp() &&
2181 (Tokens->peekNextToken()->isAccessSpecifierKeyword() ||
2182 (Previous && Previous->isAccessSpecifierKeyword()))) {
2183 nextToken();
2184 } else {
2185 parseNew();
2186 }
2187 break;
2188 case tok::kw_switch:
2189 if (Style.isJava())
2190 parseSwitch(/*IsExpr=*/true);
2191 else
2192 nextToken();
2193 break;
2194 case tok::kw_case:
2195 // Proto: there are no switch/case statements.
2196 if (Style.Language == FormatStyle::LK_Proto) {
2197 nextToken();
2198 return;
2199 }
2200 // In Verilog switch is called case.
2201 if (Style.isVerilog()) {
2202 parseBlock();
2203 addUnwrappedLine();
2204 return;
2205 }
2206 if (Style.isJavaScript() && Line->MustBeDeclaration) {
2207 // 'case: string' field declaration.
2208 nextToken();
2209 break;
2210 }
2211 parseCaseLabel();
2212 break;
2213 case tok::kw_default:
2214 nextToken();
2215 if (Style.isVerilog()) {
2216 if (FormatTok->is(tok::colon)) {
2217 // The label will be handled in the next iteration.
2218 break;
2219 }
2220 if (FormatTok->is(Keywords.kw_clocking)) {
2221 // A default clocking block.
2222 parseBlock();
2223 addUnwrappedLine();
2224 return;
2225 }
2226 parseVerilogCaseLabel();
2227 return;
2228 }
2229 break;
2230 case tok::colon:
2231 nextToken();
2232 if (Style.isVerilog()) {
2233 parseVerilogCaseLabel();
2234 return;
2235 }
2236 break;
2237 case tok::greater:
2238 nextToken();
2239 if (FormatTok->is(tok::l_brace))
2240 FormatTok->Previous->setFinalizedType(TT_TemplateCloser);
2241 break;
2242 default:
2243 nextToken();
2244 break;
2245 }
2246 }
2247}
2248
2249bool UnwrappedLineParser::tryToParsePropertyAccessor() {
2250 assert(FormatTok->is(tok::l_brace));
2251 if (!Style.isCSharp())
2252 return false;
2253 // See if it's a property accessor.
2254 if (!FormatTok->Previous || FormatTok->Previous->isNot(tok::identifier))
2255 return false;
2256
2257 // See if we are inside a property accessor.
2258 //
2259 // Record the current tokenPosition so that we can advance and
2260 // reset the current token. `Next` is not set yet so we need
2261 // another way to advance along the token stream.
2262 unsigned int StoredPosition = Tokens->getPosition();
2263 FormatToken *Tok = Tokens->getNextToken();
2264
2265 // A trivial property accessor is of the form:
2266 // { [ACCESS_SPECIFIER] [get]; [ACCESS_SPECIFIER] [set|init] }
2267 // Track these as they do not require line breaks to be introduced.
2268 bool HasSpecialAccessor = false;
2269 bool IsTrivialPropertyAccessor = true;
2270 bool HasAttribute = false;
2271 while (!eof()) {
2272 if (const bool IsAccessorKeyword =
2273 Tok->isOneOf(Keywords.kw_get, Keywords.kw_init, Keywords.kw_set);
2274 IsAccessorKeyword || Tok->isAccessSpecifierKeyword() ||
2275 Tok->isOneOf(tok::l_square, tok::semi, Keywords.kw_internal)) {
2276 if (IsAccessorKeyword)
2277 HasSpecialAccessor = true;
2278 else if (Tok->is(tok::l_square))
2279 HasAttribute = true;
2280 Tok = Tokens->getNextToken();
2281 continue;
2282 }
2283 if (Tok->isNot(tok::r_brace))
2284 IsTrivialPropertyAccessor = false;
2285 break;
2286 }
2287
2288 if (!HasSpecialAccessor || HasAttribute) {
2289 Tokens->setPosition(StoredPosition);
2290 return false;
2291 }
2292
2293 // Try to parse the property accessor:
2294 // https://docs.microsoft.com/en-us/dotnet/csharp/programming-guide/classes-and-structs/properties
2295 Tokens->setPosition(StoredPosition);
2296 if (!IsTrivialPropertyAccessor && Style.BraceWrapping.AfterFunction)
2297 addUnwrappedLine();
2298 nextToken();
2299 do {
2300 switch (FormatTok->Tok.getKind()) {
2301 case tok::r_brace:
2302 nextToken();
2303 if (FormatTok->is(tok::equal)) {
2304 while (!eof() && FormatTok->isNot(tok::semi))
2305 nextToken();
2306 nextToken();
2307 }
2308 addUnwrappedLine();
2309 return true;
2310 case tok::l_brace:
2311 ++Line->Level;
2312 parseBlock(/*MustBeDeclaration=*/true);
2313 addUnwrappedLine();
2314 --Line->Level;
2315 break;
2316 case tok::equal:
2317 if (FormatTok->is(TT_FatArrow)) {
2318 ++Line->Level;
2319 do {
2320 nextToken();
2321 } while (!eof() && FormatTok->isNot(tok::semi));
2322 nextToken();
2323 addUnwrappedLine();
2324 --Line->Level;
2325 break;
2326 }
2327 nextToken();
2328 break;
2329 default:
2330 if (FormatTok->isOneOf(Keywords.kw_get, Keywords.kw_init,
2331 Keywords.kw_set) &&
2332 !IsTrivialPropertyAccessor) {
2333 // Non-trivial get/set needs to be on its own line.
2334 addUnwrappedLine();
2335 }
2336 nextToken();
2337 }
2338 } while (!eof());
2339
2340 // Unreachable for well-formed code (paired '{' and '}').
2341 return true;
2342}
2343
2344bool UnwrappedLineParser::tryToParseLambda() {
2345 assert(FormatTok->is(tok::l_square));
2346 if (!IsCpp) {
2347 nextToken();
2348 return false;
2349 }
2350 FormatToken &LSquare = *FormatTok;
2351 if (!tryToParseLambdaIntroducer())
2352 return false;
2353
2354 FormatToken *Arrow = nullptr;
2355 bool InTemplateParameterList = false;
2356
2357 while (FormatTok->isNot(tok::l_brace)) {
2358 if (FormatTok->isTypeName(LangOpts) || FormatTok->isAttribute()) {
2359 nextToken();
2360 continue;
2361 }
2362 switch (FormatTok->Tok.getKind()) {
2363 case tok::l_brace:
2364 break;
2365 case tok::l_paren:
2366 parseParens(/*AmpAmpTokenType=*/TT_PointerOrReference);
2367 break;
2368 case tok::l_square:
2369 parseSquare();
2370 break;
2371 case tok::less:
2372 assert(FormatTok->Previous);
2373 if (FormatTok->Previous->is(tok::r_square))
2374 InTemplateParameterList = true;
2375 nextToken();
2376 break;
2377 case tok::kw_auto:
2378 case tok::kw_class:
2379 case tok::kw_struct:
2380 case tok::kw_union:
2381 case tok::kw_template:
2382 case tok::kw_typename:
2383 case tok::amp:
2384 case tok::star:
2385 case tok::kw_const:
2386 case tok::kw_constexpr:
2387 case tok::kw_consteval:
2388 case tok::comma:
2389 case tok::greater:
2390 case tok::identifier:
2391 case tok::numeric_constant:
2392 case tok::coloncolon:
2393 case tok::kw_mutable:
2394 case tok::kw_noexcept:
2395 case tok::kw_static:
2396 nextToken();
2397 break;
2398 // Specialization of a template with an integer parameter can contain
2399 // arithmetic, logical, comparison and ternary operators.
2400 //
2401 // FIXME: This also accepts sequences of operators that are not in the scope
2402 // of a template argument list.
2403 //
2404 // In a C++ lambda a template type can only occur after an arrow. We use
2405 // this as an heuristic to distinguish between Objective-C expressions
2406 // followed by an `a->b` expression, such as:
2407 // ([obj func:arg] + a->b)
2408 // Otherwise the code below would parse as a lambda.
2409 case tok::plus:
2410 case tok::minus:
2411 case tok::exclaim:
2412 case tok::tilde:
2413 case tok::slash:
2414 case tok::percent:
2415 case tok::lessless:
2416 case tok::pipe:
2417 case tok::pipepipe:
2418 case tok::ampamp:
2419 case tok::caret:
2420 case tok::equalequal:
2421 case tok::exclaimequal:
2422 case tok::greaterequal:
2423 case tok::lessequal:
2424 case tok::question:
2425 case tok::colon:
2426 case tok::ellipsis:
2427 case tok::kw_true:
2428 case tok::kw_false:
2429 if (Arrow || InTemplateParameterList) {
2430 nextToken();
2431 break;
2432 }
2433 return true;
2434 case tok::arrow:
2435 Arrow = FormatTok;
2436 nextToken();
2437 break;
2438 case tok::kw_requires:
2439 parseRequiresClause();
2440 break;
2441 case tok::equal:
2442 if (!InTemplateParameterList)
2443 return true;
2444 nextToken();
2445 break;
2446 default:
2447 return true;
2448 }
2449 }
2450
2451 FormatTok->setFinalizedType(TT_LambdaLBrace);
2452 LSquare.setFinalizedType(TT_LambdaLSquare);
2453
2454 if (Arrow)
2455 Arrow->setFinalizedType(TT_LambdaArrow);
2456
2457 NestedLambdas.push_back(Line->SeenDecltypeAuto);
2458 parseChildBlock();
2459 assert(!NestedLambdas.empty());
2460 NestedLambdas.pop_back();
2461
2462 return true;
2463}
2464
2465bool UnwrappedLineParser::tryToParseLambdaIntroducer() {
2466 const FormatToken *Previous = FormatTok->Previous;
2467 const FormatToken *LeftSquare = FormatTok;
2468 nextToken();
2469 if (Previous) {
2470 const auto *PrevPrev = Previous->getPreviousNonComment();
2471 if (Previous->is(tok::star) && PrevPrev && PrevPrev->isTypeName(LangOpts))
2472 return false;
2473 if (Previous->closesScope()) {
2474 // Not a potential C-style cast.
2475 if (Previous->isNot(tok::r_paren))
2476 return false;
2477 // Lambdas can be cast to function types only, e.g. `std::function<int()>`
2478 // and `int (*)()`.
2479 if (!PrevPrev || PrevPrev->isNoneOf(tok::greater, tok::r_paren))
2480 return false;
2481 }
2482 if (Previous && Previous->Tok.getIdentifierInfo() &&
2483 Previous->isNoneOf(tok::kw_return, tok::kw_co_await, tok::kw_co_yield,
2484 tok::kw_co_return)) {
2485 return false;
2486 }
2487 }
2488 if (LeftSquare->isCppStructuredBinding(IsCpp))
2489 return false;
2490 if (FormatTok->is(tok::l_square) || tok::isLiteral(FormatTok->Tok.getKind()))
2491 return false;
2492 if (FormatTok->is(tok::r_square)) {
2493 const FormatToken *Next = Tokens->peekNextToken(/*SkipComment=*/true);
2494 if (Next->is(tok::greater))
2495 return false;
2496 }
2497 parseSquare(/*LambdaIntroducer=*/true);
2498 return true;
2499}
2500
2501void UnwrappedLineParser::tryToParseJSFunction() {
2502 assert(FormatTok->is(Keywords.kw_function));
2503 if (FormatTok->is(Keywords.kw_async))
2504 nextToken();
2505 // Consume "function".
2506 nextToken();
2507
2508 // Consume * (generator function). Treat it like C++'s overloaded operators.
2509 if (FormatTok->is(tok::star)) {
2510 FormatTok->setFinalizedType(TT_OverloadedOperator);
2511 nextToken();
2512 }
2513
2514 // Consume function name.
2515 if (FormatTok->is(tok::identifier))
2516 nextToken();
2517
2518 if (FormatTok->isNot(tok::l_paren))
2519 return;
2520
2521 // Parse formal parameter list.
2522 parseParens();
2523
2524 if (FormatTok->is(tok::colon)) {
2525 // Parse a type definition.
2526 nextToken();
2527
2528 // Eat the type declaration. For braced inline object types, balance braces,
2529 // otherwise just parse until finding an l_brace for the function body.
2530 if (FormatTok->is(tok::l_brace))
2531 tryToParseBracedList();
2532 else
2533 while (FormatTok->isNoneOf(tok::l_brace, tok::semi) && !eof())
2534 nextToken();
2535 }
2536
2537 if (FormatTok->is(tok::semi))
2538 return;
2539
2540 parseChildBlock();
2541}
2542
2543bool UnwrappedLineParser::tryToParseBracedList() {
2544 if (FormatTok->is(BK_Unknown))
2545 calculateBraceTypes();
2546 assert(FormatTok->isNot(BK_Unknown));
2547 if (FormatTok->is(BK_Block))
2548 return false;
2549 nextToken();
2550 parseBracedList();
2551 return true;
2552}
2553
2554bool UnwrappedLineParser::tryToParseChildBlock() {
2555 assert(Style.isJavaScript() || Style.isCSharp());
2556 assert(FormatTok->is(TT_FatArrow));
2557 // Fat arrows (=>) have tok::TokenKind tok::equal but TokenType TT_FatArrow.
2558 // They always start an expression or a child block if followed by a curly
2559 // brace.
2560 nextToken();
2561 if (FormatTok->isNot(tok::l_brace))
2562 return false;
2563 parseChildBlock();
2564 return true;
2565}
2566
2567bool UnwrappedLineParser::parseBracedList(bool IsAngleBracket, bool IsEnum) {
2568 assert(!IsAngleBracket || !IsEnum);
2569 bool HasError = false;
2570
2571 // FIXME: Once we have an expression parser in the UnwrappedLineParser,
2572 // replace this by using parseAssignmentExpression() inside.
2573 do {
2574 if (Style.isCSharp() && FormatTok->is(TT_FatArrow) &&
2575 tryToParseChildBlock()) {
2576 continue;
2577 }
2578 if (Style.isJavaScript()) {
2579 if (FormatTok->is(Keywords.kw_function)) {
2580 tryToParseJSFunction();
2581 continue;
2582 }
2583 if (FormatTok->is(tok::l_brace)) {
2584 // Could be a method inside of a braced list `{a() { return 1; }}`.
2585 if (tryToParseBracedList())
2586 continue;
2587 parseChildBlock();
2588 }
2589 }
2590 if (FormatTok->is(IsAngleBracket ? tok::greater : tok::r_brace)) {
2591 if (IsEnum) {
2592 FormatTok->setBlockKind(BK_Block);
2593 if (!Style.AllowShortEnumsOnASingleLine)
2594 addUnwrappedLine();
2595 }
2596 nextToken();
2597 return !HasError;
2598 }
2599 switch (FormatTok->Tok.getKind()) {
2600 case tok::l_square:
2601 if (Style.isCSharp())
2602 parseSquare();
2603 else
2604 tryToParseLambda();
2605 break;
2606 case tok::l_paren:
2607 parseParens();
2608 // JavaScript can just have free standing methods and getters/setters in
2609 // object literals. Detect them by a "{" following ")".
2610 if (Style.isJavaScript()) {
2611 if (FormatTok->is(tok::l_brace))
2612 parseChildBlock();
2613 break;
2614 }
2615 break;
2616 case tok::l_brace:
2617 // Assume there are no blocks inside a braced init list apart
2618 // from the ones we explicitly parse out (like lambdas).
2619 FormatTok->setBlockKind(BK_BracedInit);
2620 if (!IsAngleBracket) {
2621 auto *Prev = FormatTok->Previous;
2622 if (Prev && Prev->is(tok::greater))
2623 Prev->setFinalizedType(TT_TemplateCloser);
2624 }
2625 nextToken();
2626 parseBracedList();
2627 break;
2628 case tok::less:
2629 nextToken();
2630 if (IsAngleBracket)
2631 parseBracedList(/*IsAngleBracket=*/true);
2632 break;
2633 case tok::semi:
2634 // JavaScript (or more precisely TypeScript) can have semicolons in braced
2635 // lists (in so-called TypeMemberLists). Thus, the semicolon cannot be
2636 // used for error recovery if we have otherwise determined that this is
2637 // a braced list.
2638 if (Style.isJavaScript()) {
2639 nextToken();
2640 break;
2641 }
2642 HasError = true;
2643 if (!IsEnum)
2644 return false;
2645 nextToken();
2646 break;
2647 case tok::comma:
2648 nextToken();
2649 if (IsEnum && !Style.AllowShortEnumsOnASingleLine)
2650 addUnwrappedLine();
2651 break;
2652 case tok::kw_requires:
2653 parseRequiresExpression();
2654 break;
2655 default:
2656 nextToken();
2657 break;
2658 }
2659 } while (!eof());
2660 return false;
2661}
2662
2663/// Parses a pair of parentheses (and everything between them).
2664/// \param StarAndAmpTokenType If different than TT_Unknown sets this type for
2665/// all (double) ampersands and stars. This applies for all nested scopes as
2666/// well, this is disabled within a (potential) template argument <>, and thus
2667/// also if we find only a <.
2668///
2669/// Returns whether there is a `=` token between the parentheses.
2670bool UnwrappedLineParser::parseParens(TokenType StarAndAmpTokenType,
2671 bool InMacroCall) {
2672 assert(FormatTok->is(tok::l_paren) && "'(' expected.");
2673 auto *LParen = FormatTok;
2674 auto *Prev = FormatTok->Previous;
2675 bool SeenComma = false;
2676 bool SeenEqual = false;
2677 bool MightBeFoldExpr = false;
2678 unsigned ExcessLess = 0;
2679 nextToken();
2680 const bool MightBeStmtExpr = FormatTok->is(tok::l_brace);
2681 if (!InMacroCall && Prev && Prev->is(TT_FunctionLikeMacro))
2682 InMacroCall = true;
2683 do {
2684 switch (FormatTok->Tok.getKind()) {
2685 case tok::l_paren:
2686 if (parseParens(ExcessLess == 0 ? StarAndAmpTokenType : TT_Unknown,
2687 InMacroCall)) {
2688 SeenEqual = true;
2689 }
2690 if (Style.isJava() && FormatTok->is(tok::l_brace))
2691 parseChildBlock();
2692 break;
2693 case tok::r_paren: {
2694 auto *RParen = FormatTok;
2695 nextToken();
2696 if (Prev) {
2697 auto OptionalParens = [&] {
2698 if (Style.RemoveParentheses == FormatStyle::RPS_Leave ||
2699 MightBeStmtExpr || MightBeFoldExpr || SeenComma || InMacroCall ||
2700 Line->InMacroBody || RParen->getPreviousNonComment() == LParen) {
2701 return false;
2702 }
2703 const bool DoubleParens =
2704 Prev->is(tok::l_paren) && FormatTok->is(tok::r_paren);
2705 if (DoubleParens) {
2706 const auto *PrevPrev = Prev->getPreviousNonComment();
2707 const bool Excluded =
2708 PrevPrev &&
2709 (PrevPrev->isOneOf(tok::kw___attribute, tok::kw_decltype) ||
2710 (SeenEqual &&
2711 (PrevPrev->isOneOf(tok::kw_if, tok::kw_while) ||
2712 PrevPrev->endsSequence(tok::kw_constexpr, tok::kw_if))));
2713 if (!Excluded)
2714 return true;
2715 } else {
2716 const bool CommaSeparated =
2717 Prev->isOneOf(tok::l_paren, tok::comma) &&
2718 FormatTok->isOneOf(tok::comma, tok::r_paren);
2719 if (CommaSeparated &&
2720 // LParen is not preceded by ellipsis, comma.
2721 !Prev->endsSequence(tok::comma, tok::ellipsis) &&
2722 // RParen is not followed by comma, ellipsis.
2723 !(FormatTok->is(tok::comma) &&
2724 Tokens->peekNextToken()->is(tok::ellipsis))) {
2725 return true;
2726 }
2727 const bool ReturnParens =
2728 Style.RemoveParentheses == FormatStyle::RPS_ReturnStatement &&
2729 ((NestedLambdas.empty() && !IsDecltypeAutoFunction) ||
2730 (!NestedLambdas.empty() && !NestedLambdas.back())) &&
2731 Prev->isOneOf(tok::kw_return, tok::kw_co_return) &&
2732 FormatTok->is(tok::semi);
2733 if (ReturnParens)
2734 return true;
2735 }
2736 return false;
2737 };
2738 if (OptionalParens()) {
2739 LParen->Optional = true;
2740 RParen->Optional = true;
2741 } else if (Prev->is(TT_TypenameMacro)) {
2742 LParen->setFinalizedType(TT_TypeDeclarationParen);
2743 RParen->setFinalizedType(TT_TypeDeclarationParen);
2744 } else if (Prev->is(tok::greater) && RParen->Previous == LParen) {
2745 Prev->setFinalizedType(TT_TemplateCloser);
2746 } else if (FormatTok->is(tok::l_brace) && Prev->is(tok::amp) &&
2747 !Prev->Previous) {
2748 FormatTok->setBlockKind(BK_BracedInit);
2749 }
2750 }
2751 return SeenEqual;
2752 }
2753 case tok::r_brace:
2754 // A "}" inside parenthesis is an error if there wasn't a matching "{".
2755 return SeenEqual;
2756 case tok::l_square:
2757 tryToParseLambda();
2758 break;
2759 case tok::l_brace:
2760 if (!tryToParseBracedList())
2761 parseChildBlock();
2762 break;
2763 case tok::at:
2764 nextToken();
2765 if (FormatTok->is(tok::l_brace)) {
2766 nextToken();
2767 parseBracedList();
2768 }
2769 break;
2770 case tok::comma:
2771 SeenComma = true;
2772 nextToken();
2773 break;
2774 case tok::ellipsis:
2775 MightBeFoldExpr = true;
2776 nextToken();
2777 break;
2778 case tok::equal:
2779 SeenEqual = true;
2780 if (Style.isCSharp() && FormatTok->is(TT_FatArrow))
2781 tryToParseChildBlock();
2782 else
2783 nextToken();
2784 break;
2785 case tok::kw_class:
2786 if (Style.isJavaScript())
2787 parseRecord(/*ParseAsExpr=*/true);
2788 else
2789 nextToken();
2790 break;
2791 case tok::identifier:
2792 if (Style.isJavaScript() && (FormatTok->is(Keywords.kw_function)))
2793 tryToParseJSFunction();
2794 else
2795 nextToken();
2796 break;
2797 case tok::kw_switch:
2798 if (Style.isJava())
2799 parseSwitch(/*IsExpr=*/true);
2800 else
2801 nextToken();
2802 break;
2803 case tok::kw_requires:
2804 parseRequiresExpression();
2805 break;
2806 case tok::less:
2807 // We have here no clue whether this is a less, or a template opener, opt
2808 // out of the predefined StarAndAmpTokenType.
2809 ++ExcessLess;
2810 nextToken();
2811 break;
2812 case tok::greater:
2813 if (ExcessLess > 0)
2814 --ExcessLess;
2815 nextToken();
2816 break;
2817 case tok::star:
2818 case tok::amp:
2819 case tok::ampamp:
2820 if (StarAndAmpTokenType != TT_Unknown && ExcessLess == 0)
2821 FormatTok->setFinalizedType(StarAndAmpTokenType);
2822 [[fallthrough]];
2823 default:
2824 nextToken();
2825 break;
2826 }
2827 } while (!eof());
2828 return SeenEqual;
2829}
2830
2831void UnwrappedLineParser::parseSquare(bool LambdaIntroducer) {
2832 if (!LambdaIntroducer) {
2833 assert(FormatTok->is(tok::l_square) && "'[' expected.");
2834 if (tryToParseLambda())
2835 return;
2836 }
2837 do {
2838 switch (FormatTok->Tok.getKind()) {
2839 case tok::l_paren:
2840 parseParens();
2841 break;
2842 case tok::r_square:
2843 nextToken();
2844 return;
2845 case tok::r_brace:
2846 // A "}" inside parenthesis is an error if there wasn't a matching "{".
2847 return;
2848 case tok::l_square:
2849 parseSquare();
2850 break;
2851 case tok::l_brace: {
2852 if (!tryToParseBracedList())
2853 parseChildBlock();
2854 break;
2855 }
2856 case tok::at:
2857 case tok::colon:
2858 nextToken();
2859 if (FormatTok->is(tok::l_brace)) {
2860 nextToken();
2861 parseBracedList();
2862 }
2863 break;
2864 default:
2865 nextToken();
2866 break;
2867 }
2868 } while (!eof());
2869}
2870
2871void UnwrappedLineParser::keepAncestorBraces() {
2872 if (!Style.RemoveBracesLLVM)
2873 return;
2874
2875 const int MaxNestingLevels = 2;
2876 const int Size = NestedTooDeep.size();
2877 if (Size >= MaxNestingLevels)
2878 NestedTooDeep[Size - MaxNestingLevels] = true;
2879 NestedTooDeep.push_back(false);
2880}
2881
2883 for (const auto &Token : llvm::reverse(Line.Tokens))
2884 if (Token.Tok->isNot(tok::comment))
2885 return Token.Tok;
2886
2887 return nullptr;
2888}
2889
2890void UnwrappedLineParser::parseUnbracedBody(bool CheckEOF) {
2891 FormatToken *Tok = nullptr;
2892
2893 if (Style.InsertBraces && !Line->InPPDirective && !Line->Tokens.empty() &&
2894 PreprocessorDirectives.empty() && FormatTok->isNot(tok::semi)) {
2895 Tok = Style.BraceWrapping.AfterControlStatement == FormatStyle::BWACS_Never
2896 ? getLastNonComment(*Line)
2897 : Line->Tokens.back().Tok;
2898 assert(Tok);
2899 if (Tok->BraceCount < 0) {
2900 assert(Tok->BraceCount == -1);
2901 Tok = nullptr;
2902 } else {
2903 Tok->BraceCount = -1;
2904 }
2905 }
2906
2907 addUnwrappedLine();
2908 ++Line->Level;
2909 ++Line->UnbracedBodyLevel;
2910 parseStructuralElement();
2911 --Line->UnbracedBodyLevel;
2912
2913 if (Tok) {
2914 assert(!Line->InPPDirective);
2915 Tok = nullptr;
2916 for (const auto &L : llvm::reverse(*CurrentLines)) {
2917 if (!L.InPPDirective && getLastNonComment(L)) {
2918 Tok = L.Tokens.back().Tok;
2919 break;
2920 }
2921 }
2922 assert(Tok);
2923 ++Tok->BraceCount;
2924 }
2925
2926 if (CheckEOF && eof())
2927 addUnwrappedLine();
2928
2929 --Line->Level;
2930}
2931
2932static void markOptionalBraces(FormatToken *LeftBrace) {
2933 if (!LeftBrace)
2934 return;
2935
2936 assert(LeftBrace->is(tok::l_brace));
2937
2938 FormatToken *RightBrace = LeftBrace->MatchingParen;
2939 if (!RightBrace) {
2940 assert(!LeftBrace->Optional);
2941 return;
2942 }
2943
2944 assert(RightBrace->is(tok::r_brace));
2945 assert(RightBrace->MatchingParen == LeftBrace);
2946 assert(LeftBrace->Optional == RightBrace->Optional);
2947
2948 LeftBrace->Optional = true;
2949 RightBrace->Optional = true;
2950}
2951
2952void UnwrappedLineParser::handleAttributes() {
2953 // Handle AttributeMacro, e.g. `if (x) UNLIKELY`.
2954 if (FormatTok->isAttribute())
2955 nextToken();
2956 else if (FormatTok->is(tok::l_square))
2957 handleCppAttributes();
2958}
2959
2960bool UnwrappedLineParser::handleCppAttributes() {
2961 // Handle [[likely]] / [[unlikely]] attributes.
2962 assert(FormatTok->is(tok::l_square));
2963 if (!tryToParseSimpleAttribute())
2964 return false;
2965 parseSquare();
2966 return true;
2967}
2968
2969/// Returns whether \c Tok begins a block.
2970bool UnwrappedLineParser::isBlockBegin(const FormatToken &Tok) const {
2971 // FIXME: rename the function or make
2972 // Tok.isOneOf(tok::l_brace, TT_MacroBlockBegin) work.
2973 return Style.isVerilog() ? Keywords.isVerilogBegin(Tok)
2974 : Tok.is(tok::l_brace);
2975}
2976
2977FormatToken *UnwrappedLineParser::parseIfThenElse(IfStmtKind *IfKind,
2978 bool KeepBraces,
2979 bool IsVerilogAssert) {
2980 assert((FormatTok->is(tok::kw_if) ||
2981 (Style.isVerilog() &&
2982 FormatTok->isOneOf(tok::kw_restrict, Keywords.kw_assert,
2983 Keywords.kw_assume, Keywords.kw_cover))) &&
2984 "'if' expected");
2985 nextToken();
2986
2987 if (IsVerilogAssert) {
2988 // Handle `assert #0` and `assert final`.
2989 if (FormatTok->is(Keywords.kw_verilogHash)) {
2990 nextToken();
2991 if (FormatTok->is(tok::numeric_constant))
2992 nextToken();
2993 } else if (FormatTok->isOneOf(Keywords.kw_final, Keywords.kw_property,
2994 Keywords.kw_sequence)) {
2995 nextToken();
2996 }
2997 }
2998
2999 // TableGen's if statement has the form of `if <cond> then { ... }`.
3000 if (Style.isTableGen()) {
3001 while (!eof() && FormatTok->isNot(Keywords.kw_then)) {
3002 // Simply skip until then. This range only contains a value.
3003 nextToken();
3004 }
3005 }
3006
3007 // Handle `if !consteval`.
3008 if (FormatTok->is(tok::exclaim))
3009 nextToken();
3010
3011 bool KeepIfBraces = true;
3012 if (FormatTok->is(tok::kw_consteval)) {
3013 nextToken();
3014 } else {
3015 KeepIfBraces = !Style.RemoveBracesLLVM || KeepBraces;
3016 if (FormatTok->isOneOf(tok::kw_constexpr, tok::identifier))
3017 nextToken();
3018 if (FormatTok->is(tok::l_paren)) {
3019 FormatTok->setFinalizedType(TT_ConditionLParen);
3020 parseParens();
3021 }
3022 }
3023 handleAttributes();
3024 // The then action is optional in Verilog assert statements.
3025 if (IsVerilogAssert && FormatTok->is(tok::semi)) {
3026 nextToken();
3027 addUnwrappedLine();
3028 return nullptr;
3029 }
3030
3031 bool NeedsUnwrappedLine = false;
3032 keepAncestorBraces();
3033
3034 FormatToken *IfLeftBrace = nullptr;
3035 IfStmtKind IfBlockKind = IfStmtKind::NotIf;
3036
3037 if (isBlockBegin(*FormatTok)) {
3038 FormatTok->setFinalizedType(TT_ControlStatementLBrace);
3039 IfLeftBrace = FormatTok;
3040 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3041 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
3042 /*MunchSemi=*/true, KeepIfBraces, &IfBlockKind);
3043 setPreviousRBraceType(TT_ControlStatementRBrace);
3044 if (Style.BraceWrapping.BeforeElse)
3045 addUnwrappedLine();
3046 else
3047 NeedsUnwrappedLine = true;
3048 } else if (IsVerilogAssert && FormatTok->is(tok::kw_else)) {
3049 addUnwrappedLine();
3050 } else {
3051 parseUnbracedBody();
3052 }
3053
3054 if (Style.RemoveBracesLLVM) {
3055 assert(!NestedTooDeep.empty());
3056 KeepIfBraces = KeepIfBraces ||
3057 (IfLeftBrace && !IfLeftBrace->MatchingParen) ||
3058 NestedTooDeep.back() || IfBlockKind == IfStmtKind::IfOnly ||
3059 IfBlockKind == IfStmtKind::IfElseIf;
3060 }
3061
3062 bool KeepElseBraces = KeepIfBraces;
3063 FormatToken *ElseLeftBrace = nullptr;
3064 IfStmtKind Kind = IfStmtKind::IfOnly;
3065
3066 if (FormatTok->is(tok::kw_else)) {
3067 if (Style.RemoveBracesLLVM) {
3068 NestedTooDeep.back() = false;
3069 Kind = IfStmtKind::IfElse;
3070 }
3071 nextToken();
3072 handleAttributes();
3073 if (isBlockBegin(*FormatTok)) {
3074 const bool FollowedByIf = Tokens->peekNextToken()->is(tok::kw_if);
3075 FormatTok->setFinalizedType(TT_ElseLBrace);
3076 ElseLeftBrace = FormatTok;
3077 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3078 IfStmtKind ElseBlockKind = IfStmtKind::NotIf;
3079 FormatToken *IfLBrace =
3080 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
3081 /*MunchSemi=*/true, KeepElseBraces, &ElseBlockKind);
3082 setPreviousRBraceType(TT_ElseRBrace);
3083 if (FormatTok->is(tok::kw_else)) {
3084 KeepElseBraces = KeepElseBraces ||
3085 ElseBlockKind == IfStmtKind::IfOnly ||
3086 ElseBlockKind == IfStmtKind::IfElseIf;
3087 } else if (FollowedByIf && IfLBrace && !IfLBrace->Optional) {
3088 KeepElseBraces = true;
3089 assert(ElseLeftBrace->MatchingParen);
3090 markOptionalBraces(ElseLeftBrace);
3091 }
3092 addUnwrappedLine();
3093 } else if (!IsVerilogAssert && FormatTok->is(tok::kw_if)) {
3094 const FormatToken *Previous = Tokens->getPreviousToken();
3095 assert(Previous);
3096 const bool IsPrecededByComment = Previous->is(tok::comment);
3097 if (IsPrecededByComment) {
3098 addUnwrappedLine();
3099 ++Line->Level;
3100 }
3101 bool TooDeep = true;
3102 if (Style.RemoveBracesLLVM) {
3103 Kind = IfStmtKind::IfElseIf;
3104 TooDeep = NestedTooDeep.pop_back_val();
3105 }
3106 ElseLeftBrace = parseIfThenElse(/*IfKind=*/nullptr, KeepIfBraces);
3107 if (Style.RemoveBracesLLVM)
3108 NestedTooDeep.push_back(TooDeep);
3109 if (IsPrecededByComment)
3110 --Line->Level;
3111 } else {
3112 parseUnbracedBody(/*CheckEOF=*/true);
3113 }
3114 } else {
3115 KeepIfBraces = KeepIfBraces || IfBlockKind == IfStmtKind::IfElse;
3116 if (NeedsUnwrappedLine)
3117 addUnwrappedLine();
3118 }
3119
3120 if (!Style.RemoveBracesLLVM)
3121 return nullptr;
3122
3123 assert(!NestedTooDeep.empty());
3124 KeepElseBraces = KeepElseBraces ||
3125 (ElseLeftBrace && !ElseLeftBrace->MatchingParen) ||
3126 NestedTooDeep.back();
3127
3128 NestedTooDeep.pop_back();
3129
3130 if (!KeepIfBraces && !KeepElseBraces) {
3131 markOptionalBraces(IfLeftBrace);
3132 markOptionalBraces(ElseLeftBrace);
3133 } else if (IfLeftBrace) {
3134 FormatToken *IfRightBrace = IfLeftBrace->MatchingParen;
3135 if (IfRightBrace) {
3136 assert(IfRightBrace->MatchingParen == IfLeftBrace);
3137 assert(!IfLeftBrace->Optional);
3138 assert(!IfRightBrace->Optional);
3139 IfLeftBrace->MatchingParen = nullptr;
3140 IfRightBrace->MatchingParen = nullptr;
3141 }
3142 }
3143
3144 if (IfKind)
3145 *IfKind = Kind;
3146
3147 return IfLeftBrace;
3148}
3149
3150void UnwrappedLineParser::parseTryCatch() {
3151 assert(FormatTok->isOneOf(tok::kw_try, tok::kw___try) && "'try' expected");
3152 nextToken();
3153 bool NeedsUnwrappedLine = false;
3154 bool HasCtorInitializer = false;
3155 if (FormatTok->is(tok::colon)) {
3156 auto *Colon = FormatTok;
3157 // We are in a function try block, what comes is an initializer list.
3158 nextToken();
3159 if (FormatTok->is(tok::identifier)) {
3160 HasCtorInitializer = true;
3161 Colon->setFinalizedType(TT_CtorInitializerColon);
3162 }
3163
3164 // In case identifiers were removed by clang-tidy, what might follow is
3165 // multiple commas in sequence - before the first identifier.
3166 while (FormatTok->is(tok::comma))
3167 nextToken();
3168
3169 while (FormatTok->is(tok::identifier)) {
3170 nextToken();
3171 if (FormatTok->is(tok::l_paren)) {
3172 parseParens();
3173 } else if (FormatTok->is(tok::l_brace)) {
3174 nextToken();
3175 parseBracedList();
3176 }
3177
3178 // In case identifiers were removed by clang-tidy, what might follow is
3179 // multiple commas in sequence - after the first identifier.
3180 while (FormatTok->is(tok::comma))
3181 nextToken();
3182 }
3183 }
3184 // Parse try with resource.
3185 if (Style.isJava() && FormatTok->is(tok::l_paren))
3186 parseParens();
3187
3188 keepAncestorBraces();
3189
3190 if (FormatTok->is(tok::l_brace)) {
3191 if (HasCtorInitializer)
3192 FormatTok->setFinalizedType(TT_FunctionLBrace);
3193 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3194 parseBlock();
3195 if (Style.BraceWrapping.BeforeCatch)
3196 addUnwrappedLine();
3197 else
3198 NeedsUnwrappedLine = true;
3199 } else if (FormatTok->isNot(tok::kw_catch)) {
3200 // The C++ standard requires a compound-statement after a try.
3201 // If there's none, we try to assume there's a structuralElement
3202 // and try to continue.
3203 addUnwrappedLine();
3204 ++Line->Level;
3205 parseStructuralElement();
3206 --Line->Level;
3207 }
3208 for (bool SeenCatch = false;;) {
3209 if (FormatTok->is(tok::at))
3210 nextToken();
3211 if (FormatTok->isNoneOf(tok::kw_catch, Keywords.kw___except,
3212 tok::kw___finally, tok::objc_catch,
3213 tok::objc_finally) &&
3214 !((Style.isJava() || Style.isJavaScript()) &&
3215 FormatTok->is(Keywords.kw_finally))) {
3216 break;
3217 }
3218 if (FormatTok->is(tok::kw_catch))
3219 SeenCatch = true;
3220 nextToken();
3221 while (FormatTok->isNot(tok::l_brace)) {
3222 if (FormatTok->is(tok::l_paren)) {
3223 parseParens();
3224 continue;
3225 }
3226 if (FormatTok->isOneOf(tok::semi, tok::r_brace) || eof()) {
3227 if (Style.RemoveBracesLLVM)
3228 NestedTooDeep.pop_back();
3229 return;
3230 }
3231 nextToken();
3232 }
3233 if (SeenCatch) {
3234 FormatTok->setFinalizedType(TT_ControlStatementLBrace);
3235 SeenCatch = false;
3236 }
3237 NeedsUnwrappedLine = false;
3238 Line->MustBeDeclaration = false;
3239 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3240 parseBlock();
3241 if (Style.BraceWrapping.BeforeCatch)
3242 addUnwrappedLine();
3243 else
3244 NeedsUnwrappedLine = true;
3245 }
3246
3247 if (Style.RemoveBracesLLVM)
3248 NestedTooDeep.pop_back();
3249
3250 if (NeedsUnwrappedLine)
3251 addUnwrappedLine();
3252}
3253
3254void UnwrappedLineParser::parseNamespaceOrExportBlock(unsigned AddLevels) {
3255 bool ManageWhitesmithsBraces =
3256 AddLevels == 0u && Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths;
3257
3258 // If we're in Whitesmiths mode, indent the brace if we're not indenting
3259 // the whole block.
3260 if (ManageWhitesmithsBraces)
3261 ++Line->Level;
3262
3263 // Munch the semicolon after the block. This is more common than one would
3264 // think. Putting the semicolon into its own line is very ugly.
3265 parseBlock(/*MustBeDeclaration=*/true, AddLevels, /*MunchSemi=*/true,
3266 /*KeepBraces=*/true, /*IfKind=*/nullptr, ManageWhitesmithsBraces);
3267
3268 addUnwrappedLine(AddLevels > 0 ? LineLevel::Remove : LineLevel::Keep);
3269
3270 if (ManageWhitesmithsBraces)
3271 --Line->Level;
3272}
3273
3274void UnwrappedLineParser::parseNamespace() {
3275 assert(FormatTok->isOneOf(tok::kw_namespace, TT_NamespaceMacro) &&
3276 "'namespace' expected");
3277
3278 const FormatToken &InitialToken = *FormatTok;
3279 nextToken();
3280 if (InitialToken.is(TT_NamespaceMacro)) {
3281 parseParens();
3282 } else {
3283 while (FormatTok->isOneOf(tok::identifier, tok::coloncolon, tok::kw_inline,
3284 tok::l_square, tok::period, tok::l_paren) ||
3285 (Style.isCSharp() && FormatTok->is(tok::kw_union))) {
3286 if (FormatTok->is(tok::l_square))
3287 parseSquare();
3288 else if (FormatTok->is(tok::l_paren))
3289 parseParens();
3290 else
3291 nextToken();
3292 }
3293 }
3294 if (FormatTok->is(tok::l_brace)) {
3295 FormatTok->setFinalizedType(TT_NamespaceLBrace);
3296
3297 if (ShouldBreakBeforeBrace(Style, InitialToken,
3298 Tokens->peekNextToken()->is(tok::r_brace))) {
3299 addUnwrappedLine();
3300 }
3301
3302 unsigned AddLevels =
3303 Style.NamespaceIndentation == FormatStyle::NI_All ||
3304 (Style.NamespaceIndentation == FormatStyle::NI_Inner &&
3305 DeclarationScopeStack.size() > 1)
3306 ? 1u
3307 : 0u;
3308 parseNamespaceOrExportBlock(AddLevels);
3309 }
3310 // FIXME: Add error handling.
3311}
3312
3313void UnwrappedLineParser::parseCppExportBlock() {
3314 if (FormatTok->is(tok::l_brace)) {
3315 FormatTok->setFinalizedType(TT_ExportLBrace);
3316 if (Style.BraceWrapping.AfterExportBlock)
3317 addUnwrappedLine();
3318 }
3319 parseNamespaceOrExportBlock(/*AddLevels=*/Style.IndentExportBlock ? 1 : 0);
3320}
3321
3322void UnwrappedLineParser::parseNew() {
3323 assert(FormatTok->is(tok::kw_new) && "'new' expected");
3324 nextToken();
3325
3326 if (Style.isCSharp()) {
3327 do {
3328 // Handle constructor invocation, e.g. `new(field: value)`.
3329 if (FormatTok->is(tok::l_paren))
3330 parseParens();
3331
3332 // Handle array initialization syntax, e.g. `new[] {10, 20, 30}`.
3333 if (FormatTok->is(tok::l_brace))
3334 parseBracedList();
3335
3336 if (FormatTok->isOneOf(tok::semi, tok::comma))
3337 return;
3338
3339 nextToken();
3340 } while (!eof());
3341 }
3342
3343 if (!Style.isJava())
3344 return;
3345
3346 // In Java, we can parse everything up to the parens, which aren't optional.
3347 do {
3348 // There should not be a ;, { or } before the new's open paren.
3349 if (FormatTok->isOneOf(tok::semi, tok::l_brace, tok::r_brace))
3350 return;
3351
3352 // Consume the parens.
3353 if (FormatTok->is(tok::l_paren)) {
3354 parseParens();
3355
3356 // If there is a class body of an anonymous class, consume that as child.
3357 if (FormatTok->is(tok::l_brace))
3358 parseChildBlock();
3359 return;
3360 }
3361 nextToken();
3362 } while (!eof());
3363}
3364
3365void UnwrappedLineParser::parseLoopBody(bool KeepBraces, bool WrapRightBrace) {
3366 keepAncestorBraces();
3367
3368 if (isBlockBegin(*FormatTok)) {
3369 FormatTok->setFinalizedType(TT_ControlStatementLBrace);
3370 FormatToken *LeftBrace = FormatTok;
3371 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3372 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
3373 /*MunchSemi=*/true, KeepBraces);
3374 setPreviousRBraceType(TT_ControlStatementRBrace);
3375 if (!KeepBraces) {
3376 assert(!NestedTooDeep.empty());
3377 if (!NestedTooDeep.back())
3378 markOptionalBraces(LeftBrace);
3379 }
3380 if (WrapRightBrace)
3381 addUnwrappedLine();
3382 } else {
3383 parseUnbracedBody();
3384 }
3385
3386 if (!KeepBraces)
3387 NestedTooDeep.pop_back();
3388}
3389
3390void UnwrappedLineParser::parseForOrWhileLoop(bool HasParens) {
3391 assert((FormatTok->isOneOf(tok::kw_for, tok::kw_while, TT_ForEachMacro) ||
3392 (Style.isVerilog() &&
3393 FormatTok->isOneOf(Keywords.kw_always, Keywords.kw_always_comb,
3394 Keywords.kw_always_ff, Keywords.kw_always_latch,
3395 Keywords.kw_final, Keywords.kw_initial,
3396 Keywords.kw_foreach, Keywords.kw_forever,
3397 Keywords.kw_repeat))) &&
3398 "'for', 'while' or foreach macro expected");
3399 const bool KeepBraces = !Style.RemoveBracesLLVM ||
3400 FormatTok->isNoneOf(tok::kw_for, tok::kw_while);
3401
3402 nextToken();
3403 // JS' for await ( ...
3404 if (Style.isJavaScript() && FormatTok->is(Keywords.kw_await))
3405 nextToken();
3406 if (IsCpp && FormatTok->is(tok::kw_co_await))
3407 nextToken();
3408 if (HasParens && FormatTok->is(tok::l_paren)) {
3409 // The type is only set for Verilog basically because we were afraid to
3410 // change the existing behavior for loops. See the discussion on D121756 for
3411 // details.
3412 if (Style.isVerilog())
3413 FormatTok->setFinalizedType(TT_ConditionLParen);
3414 parseParens();
3415 }
3416
3417 if (Style.isVerilog()) {
3418 // Event control.
3419 parseVerilogSensitivityList();
3420 } else if (Style.AllowShortLoopsOnASingleLine && FormatTok->is(tok::semi) &&
3421 Tokens->getPreviousToken()->is(tok::r_paren)) {
3422 nextToken();
3423 addUnwrappedLine();
3424 return;
3425 }
3426
3427 handleAttributes();
3428 parseLoopBody(KeepBraces, /*WrapRightBrace=*/true);
3429}
3430
3431void UnwrappedLineParser::parseDoWhile() {
3432 assert(FormatTok->is(tok::kw_do) && "'do' expected");
3433 nextToken();
3434
3435 parseLoopBody(/*KeepBraces=*/true, Style.BraceWrapping.BeforeWhile);
3436
3437 // FIXME: Add error handling.
3438 if (FormatTok->isNot(tok::kw_while)) {
3439 addUnwrappedLine();
3440 return;
3441 }
3442
3443 FormatTok->setFinalizedType(TT_DoWhile);
3444
3445 // If in Whitesmiths mode, the line with the while() needs to be indented
3446 // to the same level as the block.
3447 if (Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths)
3448 ++Line->Level;
3449
3450 nextToken();
3451 parseStructuralElement();
3452}
3453
3454void UnwrappedLineParser::parseLabel(bool IsGotoLabel) {
3455 nextToken();
3456
3457 const auto IndentGotoLabel = Style.IndentGotoLabels;
3458 const auto OldLineLevel = Line->Level;
3459 auto &Level = Line->Level;
3460
3461 if (IsGotoLabel && IndentGotoLabel == FormatStyle::IGLS_NoIndent)
3462 Level = 0;
3463
3464 if (!IsGotoLabel || IndentGotoLabel == FormatStyle::IGLS_OuterIndent) {
3465 if (OldLineLevel > 1 || (!Line->InPPDirective && OldLineLevel > 0))
3466 --Level;
3467 }
3468
3469 if (!IsGotoLabel && !Style.IndentCaseBlocks &&
3470 CommentsBeforeNextToken.empty() && FormatTok->is(tok::l_brace)) {
3471 CompoundStatementIndenter Indenter(this, Level,
3472 Style.BraceWrapping.AfterCaseLabel,
3473 Style.BraceWrapping.IndentBraces);
3474 parseBlock();
3475 if (FormatTok->is(tok::kw_break)) {
3476 if (Style.BraceWrapping.AfterControlStatement ==
3478 addUnwrappedLine();
3479 if (!Style.IndentCaseBlocks &&
3480 Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths) {
3481 ++Level;
3482 }
3483 }
3484 parseStructuralElement();
3485 }
3486 addUnwrappedLine();
3487 } else {
3488 if (FormatTok->is(tok::semi))
3489 nextToken();
3490 addUnwrappedLine();
3491 }
3492
3493 Level = OldLineLevel;
3494
3495 if (FormatTok->isNot(tok::l_brace)) {
3496 parseStructuralElement();
3497 addUnwrappedLine();
3498 }
3499}
3500
3501void UnwrappedLineParser::parseCaseLabel() {
3502 assert(FormatTok->is(tok::kw_case) && "'case' expected");
3503 auto *Case = FormatTok;
3504
3505 // FIXME: fix handling of complex expressions here.
3506 do {
3507 nextToken();
3508 if (FormatTok->is(tok::colon)) {
3509 FormatTok->setFinalizedType(TT_CaseLabelColon);
3510 break;
3511 }
3512 if (Style.isJava() && FormatTok->is(tok::arrow)) {
3513 FormatTok->setFinalizedType(TT_CaseLabelArrow);
3514 Case->setFinalizedType(TT_SwitchExpressionLabel);
3515 break;
3516 }
3517 } while (!eof());
3518 parseLabel();
3519}
3520
3521void UnwrappedLineParser::parseSwitch(bool IsExpr) {
3522 assert(FormatTok->is(tok::kw_switch) && "'switch' expected");
3523 nextToken();
3524 if (FormatTok->is(tok::l_paren))
3525 parseParens();
3526
3527 keepAncestorBraces();
3528
3529 if (FormatTok->is(tok::l_brace)) {
3530 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3531 FormatTok->setFinalizedType(IsExpr ? TT_SwitchExpressionLBrace
3532 : TT_ControlStatementLBrace);
3533 if (IsExpr)
3534 parseChildBlock();
3535 else
3536 parseBlock();
3537 setPreviousRBraceType(TT_ControlStatementRBrace);
3538 if (!IsExpr)
3539 addUnwrappedLine();
3540 } else {
3541 addUnwrappedLine();
3542 ++Line->Level;
3543 parseStructuralElement();
3544 --Line->Level;
3545 }
3546
3547 if (Style.RemoveBracesLLVM)
3548 NestedTooDeep.pop_back();
3549}
3550
3551void UnwrappedLineParser::parseAccessSpecifier() {
3552 nextToken();
3553 // Understand Qt's slots.
3554 if (FormatTok->isOneOf(Keywords.kw_slots, Keywords.kw_qslots))
3555 nextToken();
3556 // Otherwise, we don't know what it is, and we'd better keep the next token.
3557 if (FormatTok->is(tok::colon))
3558 nextToken();
3559 addUnwrappedLine();
3560}
3561
3562/// Parses a requires, decides if it is a clause or an expression.
3563/// \pre The current token has to be the requires keyword.
3564/// \returns true if it parsed a clause.
3565bool UnwrappedLineParser::parseRequires(bool SeenEqual) {
3566 assert(FormatTok->is(tok::kw_requires) && "'requires' expected");
3567
3568 // We try to guess if it is a requires clause, or a requires expression. For
3569 // that we first check the next token.
3570 switch (Tokens->peekNextToken(/*SkipComment=*/true)->Tok.getKind()) {
3571 case tok::l_brace:
3572 // This can only be an expression, never a clause.
3573 parseRequiresExpression();
3574 return false;
3575 case tok::l_paren:
3576 // Clauses and expression can start with a paren, it's unclear what we have.
3577 break;
3578 default:
3579 // All other tokens can only be a clause.
3580 parseRequiresClause();
3581 return true;
3582 }
3583
3584 // Looking forward we would have to decide if there are function declaration
3585 // like arguments to the requires expression:
3586 // requires (T t) {
3587 // Or there is a constraint expression for the requires clause:
3588 // requires (C<T> && ...
3589
3590 // But first let's look behind.
3591 auto *PreviousNonComment = FormatTok->getPreviousNonComment();
3592
3593 if (!PreviousNonComment ||
3594 PreviousNonComment->is(TT_RequiresExpressionLBrace)) {
3595 // If there is no token, or an expression left brace, we are a requires
3596 // clause within a requires expression.
3597 parseRequiresClause();
3598 return true;
3599 }
3600
3601 switch (PreviousNonComment->Tok.getKind()) {
3602 case tok::greater:
3603 case tok::r_paren:
3604 case tok::kw_noexcept:
3605 case tok::kw_const:
3606 case tok::star:
3607 case tok::amp:
3608 // This is a requires clause.
3609 parseRequiresClause();
3610 return true;
3611 case tok::ampamp: {
3612 // This can be either:
3613 // if (... && requires (T t) ...)
3614 // Or
3615 // void member(...) && requires (C<T> ...
3616 // We check the one token before that for a const:
3617 // void member(...) const && requires (C<T> ...
3618 auto PrevPrev = PreviousNonComment->getPreviousNonComment();
3619 if ((PrevPrev && PrevPrev->is(tok::kw_const)) || !SeenEqual) {
3620 parseRequiresClause();
3621 return true;
3622 }
3623 break;
3624 }
3625 default:
3626 if (PreviousNonComment->isTypeOrIdentifier(LangOpts)) {
3627 // This is a requires clause.
3628 parseRequiresClause();
3629 return true;
3630 }
3631 // It's an expression.
3632 parseRequiresExpression();
3633 return false;
3634 }
3635
3636 // Now we look forward and try to check if the paren content is a parameter
3637 // list. The parameters can be cv-qualified and contain references or
3638 // pointers.
3639 // So we want basically to check for TYPE NAME, but TYPE can contain all kinds
3640 // of stuff: typename, const, *, &, &&, ::, identifiers.
3641
3642 unsigned StoredPosition = Tokens->getPosition();
3643 FormatToken *NextToken = Tokens->getNextToken();
3644 int Lookahead = 0;
3645 auto PeekNext = [&Lookahead, &NextToken, this] {
3646 ++Lookahead;
3647 NextToken = Tokens->getNextToken();
3648 };
3649
3650 bool FoundType = false;
3651 bool LastWasColonColon = false;
3652 int OpenAngles = 0;
3653
3654 for (; Lookahead < 50; PeekNext()) {
3655 switch (NextToken->Tok.getKind()) {
3656 case tok::kw_volatile:
3657 case tok::kw_const:
3658 case tok::comma:
3659 if (OpenAngles == 0) {
3660 FormatTok = Tokens->setPosition(StoredPosition);
3661 parseRequiresExpression();
3662 return false;
3663 }
3664 break;
3665 case tok::eof:
3666 // Break out of the loop.
3667 Lookahead = 50;
3668 break;
3669 case tok::coloncolon:
3670 LastWasColonColon = true;
3671 break;
3672 case tok::kw_decltype:
3673 case tok::identifier:
3674 if (FoundType && !LastWasColonColon && OpenAngles == 0) {
3675 FormatTok = Tokens->setPosition(StoredPosition);
3676 parseRequiresExpression();
3677 return false;
3678 }
3679 FoundType = true;
3680 LastWasColonColon = false;
3681 break;
3682 case tok::less:
3683 ++OpenAngles;
3684 break;
3685 case tok::greater:
3686 --OpenAngles;
3687 break;
3688 default:
3689 if (NextToken->isTypeName(LangOpts)) {
3690 FormatTok = Tokens->setPosition(StoredPosition);
3691 parseRequiresExpression();
3692 return false;
3693 }
3694 break;
3695 }
3696 }
3697 // This seems to be a complicated expression, just assume it's a clause.
3698 FormatTok = Tokens->setPosition(StoredPosition);
3699 parseRequiresClause();
3700 return true;
3701}
3702
3703/// Parses a requires clause.
3704/// \sa parseRequiresExpression
3705///
3706/// Returns if it either has finished parsing the clause, or it detects, that
3707/// the clause is incorrect.
3708void UnwrappedLineParser::parseRequiresClause() {
3709 assert(FormatTok->is(tok::kw_requires) && "'requires' expected");
3710
3711 // If there is no previous token, we are within a requires expression,
3712 // otherwise we will always have the template or function declaration in front
3713 // of it.
3714 bool InRequiresExpression =
3715 !FormatTok->Previous ||
3716 FormatTok->Previous->is(TT_RequiresExpressionLBrace);
3717
3718 FormatTok->setFinalizedType(InRequiresExpression
3719 ? TT_RequiresClauseInARequiresExpression
3720 : TT_RequiresClause);
3721 nextToken();
3722
3723 // NOTE: parseConstraintExpression is only ever called from this function.
3724 // It could be inlined into here.
3725 parseConstraintExpression();
3726
3727 if (!InRequiresExpression && FormatTok->Previous)
3728 FormatTok->Previous->ClosesRequiresClause = true;
3729}
3730
3731/// Parses a requires expression.
3732/// \sa parseRequiresClause
3733///
3734/// Returns if it either has finished parsing the expression, or it detects,
3735/// that the expression is incorrect.
3736void UnwrappedLineParser::parseRequiresExpression() {
3737 assert(FormatTok->is(tok::kw_requires) && "'requires' expected");
3738
3739 FormatTok->setFinalizedType(TT_RequiresExpression);
3740 nextToken();
3741
3742 if (FormatTok->is(tok::l_paren)) {
3743 FormatTok->setFinalizedType(TT_RequiresExpressionLParen);
3744 parseParens();
3745 }
3746
3747 if (FormatTok->is(tok::l_brace)) {
3748 FormatTok->setFinalizedType(TT_RequiresExpressionLBrace);
3749 parseChildBlock();
3750 }
3751}
3752
3753/// Parses a constraint expression.
3754///
3755/// This is the body of a requires clause. It returns, when the parsing is
3756/// complete, or the expression is incorrect.
3757void UnwrappedLineParser::parseConstraintExpression() {
3758 // The special handling for lambdas is needed since tryToParseLambda() eats a
3759 // token and if a requires expression is the last part of a requires clause
3760 // and followed by an attribute like [[nodiscard]] the ClosesRequiresClause is
3761 // not set on the correct token. Thus we need to be aware if we even expect a
3762 // lambda to be possible.
3763 // template <typename T> requires requires { ... } [[nodiscard]] ...;
3764 bool LambdaNextTimeAllowed = true;
3765
3766 // Within lambda declarations, it is permitted to put a requires clause after
3767 // its template parameter list, which would place the requires clause right
3768 // before the parentheses of the parameters of the lambda declaration. Thus,
3769 // we track if we expect to see grouping parentheses at all.
3770 // Without this check, `requires foo<T> (T t)` in the below example would be
3771 // seen as the whole requires clause, accidentally eating the parameters of
3772 // the lambda.
3773 // [&]<typename T> requires foo<T> (T t) { ... };
3774 bool TopLevelParensAllowed = true;
3775
3776 do {
3777 bool LambdaThisTimeAllowed = std::exchange(LambdaNextTimeAllowed, false);
3778
3779 switch (FormatTok->Tok.getKind()) {
3780 case tok::kw_requires:
3781 parseRequiresExpression();
3782 break;
3783
3784 case tok::l_paren:
3785 if (!TopLevelParensAllowed)
3786 return;
3787 parseParens(/*AmpAmpTokenType=*/TT_BinaryOperator);
3788 TopLevelParensAllowed = false;
3789 break;
3790
3791 case tok::l_square:
3792 if (!LambdaThisTimeAllowed || !tryToParseLambda())
3793 return;
3794 break;
3795
3796 case tok::kw_const:
3797 case tok::semi:
3798 case tok::kw_class:
3799 case tok::kw_struct:
3800 case tok::kw_union:
3801 return;
3802
3803 case tok::l_brace:
3804 // Potential function body.
3805 return;
3806
3807 case tok::ampamp:
3808 case tok::pipepipe:
3809 FormatTok->setFinalizedType(TT_BinaryOperator);
3810 nextToken();
3811 LambdaNextTimeAllowed = true;
3812 TopLevelParensAllowed = true;
3813 break;
3814
3815 case tok::comma:
3816 case tok::comment:
3817 LambdaNextTimeAllowed = LambdaThisTimeAllowed;
3818 nextToken();
3819 break;
3820
3821 case tok::kw_sizeof:
3822 case tok::greater:
3823 case tok::greaterequal:
3824 case tok::greatergreater:
3825 case tok::less:
3826 case tok::lessequal:
3827 case tok::lessless:
3828 case tok::equalequal:
3829 case tok::exclaim:
3830 case tok::exclaimequal:
3831 case tok::plus:
3832 case tok::minus:
3833 case tok::star:
3834 case tok::slash:
3835 LambdaNextTimeAllowed = true;
3836 TopLevelParensAllowed = true;
3837 // Just eat them.
3838 nextToken();
3839 break;
3840
3841 case tok::numeric_constant:
3842 case tok::coloncolon:
3843 case tok::kw_true:
3844 case tok::kw_false:
3845 TopLevelParensAllowed = false;
3846 // Just eat them.
3847 nextToken();
3848 break;
3849
3850 case tok::kw_static_cast:
3851 case tok::kw_const_cast:
3852 case tok::kw_reinterpret_cast:
3853 case tok::kw_dynamic_cast:
3854 nextToken();
3855 if (FormatTok->isNot(tok::less))
3856 return;
3857
3858 nextToken();
3859 parseBracedList(/*IsAngleBracket=*/true);
3860 break;
3861
3862 default:
3863 if (!FormatTok->Tok.getIdentifierInfo()) {
3864 // Identifiers are part of the default case, we check for more then
3865 // tok::identifier to handle builtin type traits.
3866 return;
3867 }
3868
3869 // We need to differentiate identifiers for a template deduction guide,
3870 // variables, or function return types (the constraint expression has
3871 // ended before that), and basically all other cases. But it's easier to
3872 // check the other way around.
3873 assert(FormatTok->Previous);
3874 switch (FormatTok->Previous->Tok.getKind()) {
3875 case tok::coloncolon: // Nested identifier.
3876 case tok::ampamp: // Start of a function or variable for the
3877 case tok::pipepipe: // constraint expression. (binary)
3878 case tok::exclaim: // The same as above, but unary.
3879 case tok::kw_requires: // Initial identifier of a requires clause.
3880 case tok::equal: // Initial identifier of a concept declaration.
3881 case tok::kw_template: // A dependent template.
3882 break;
3883 default:
3884 return;
3885 }
3886
3887 // Read identifier with optional template declaration.
3888 nextToken();
3889 if (FormatTok->is(tok::less)) {
3890 nextToken();
3891 parseBracedList(/*IsAngleBracket=*/true);
3892 }
3893 TopLevelParensAllowed = false;
3894 break;
3895 }
3896 } while (!eof());
3897}
3898
3899bool UnwrappedLineParser::parseEnum() {
3900 const FormatToken &InitialToken = *FormatTok;
3901
3902 // Won't be 'enum' for NS_ENUMs.
3903 if (FormatTok->is(tok::kw_enum))
3904 nextToken();
3905
3906 // In TypeScript, "enum" can also be used as property name, e.g. in interface
3907 // declarations. An "enum" keyword followed by a colon would be a syntax
3908 // error and thus assume it is just an identifier.
3909 if (Style.isJavaScript() && FormatTok->isOneOf(tok::colon, tok::question))
3910 return false;
3911
3912 // In protobuf, "enum" can be used as a field name.
3913 if (Style.Language == FormatStyle::LK_Proto && FormatTok->is(tok::equal))
3914 return false;
3915
3916 if (IsCpp) {
3917 // Eat up enum class ...
3918 if (FormatTok->isOneOf(tok::kw_class, tok::kw_struct))
3919 nextToken();
3920 while (FormatTok->is(tok::l_square))
3921 if (!handleCppAttributes())
3922 return false;
3923 }
3924
3925 while (FormatTok->Tok.getIdentifierInfo() ||
3926 FormatTok->isOneOf(tok::colon, tok::coloncolon, tok::less,
3927 tok::greater, tok::comma, tok::question,
3928 tok::l_square)) {
3929 if (FormatTok->is(tok::colon))
3930 FormatTok->setFinalizedType(TT_EnumUnderlyingTypeColon);
3931 if (Style.isVerilog()) {
3932 FormatTok->setFinalizedType(TT_VerilogDimensionedTypeName);
3933 nextToken();
3934 // In Verilog the base type can have dimensions.
3935 while (FormatTok->is(tok::l_square))
3936 parseSquare();
3937 } else {
3938 nextToken();
3939 }
3940 // We can have macros or attributes in between 'enum' and the enum name.
3941 if (FormatTok->is(tok::l_paren))
3942 parseParens();
3943 if (FormatTok->is(tok::identifier)) {
3944 nextToken();
3945 // If there are two identifiers in a row, this is likely an elaborate
3946 // return type. In Java, this can be "implements", etc.
3947 if (IsCpp && FormatTok->is(tok::identifier))
3948 return false;
3949 }
3950 }
3951
3952 // Just a declaration or something is wrong.
3953 if (FormatTok->isNot(tok::l_brace))
3954 return true;
3955 FormatTok->setFinalizedType(TT_EnumLBrace);
3956 FormatTok->setBlockKind(BK_Block);
3957
3958 if (Style.isJava()) {
3959 // Java enums are different.
3960 parseJavaEnumBody();
3961 return true;
3962 }
3963 if (Style.Language == FormatStyle::LK_Proto) {
3964 parseBlock(/*MustBeDeclaration=*/true);
3965 return true;
3966 }
3967
3968 const bool ManageWhitesmithsBraces =
3969 Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths;
3970
3971 if (!Style.AllowShortEnumsOnASingleLine &&
3972 ShouldBreakBeforeBrace(Style, InitialToken,
3973 Tokens->peekNextToken()->is(tok::r_brace))) {
3974 addUnwrappedLine();
3975
3976 // If we're in Whitesmiths mode, indent the brace if we're not indenting
3977 // the whole block.
3978 if (ManageWhitesmithsBraces)
3979 ++Line->Level;
3980 }
3981 // Parse enum body.
3982 nextToken();
3983 if (!Style.AllowShortEnumsOnASingleLine) {
3984 addUnwrappedLine();
3985 if (!ManageWhitesmithsBraces)
3986 ++Line->Level;
3987 }
3988 const auto OpeningLineIndex = CurrentLines->empty()
3989 ? UnwrappedLine::kInvalidIndex
3990 : CurrentLines->size() - 1;
3991 bool HasError = !parseBracedList(/*IsAngleBracket=*/false, /*IsEnum=*/true);
3992 if (!Style.AllowShortEnumsOnASingleLine && !ManageWhitesmithsBraces)
3993 --Line->Level;
3994 if (HasError) {
3995 if (FormatTok->is(tok::semi))
3996 nextToken();
3997 addUnwrappedLine();
3998 }
3999 setPreviousRBraceType(TT_EnumRBrace);
4000 if (ManageWhitesmithsBraces)
4001 Line->MatchingOpeningBlockLineIndex = OpeningLineIndex;
4002 return true;
4003
4004 // There is no addUnwrappedLine() here so that we fall through to parsing a
4005 // structural element afterwards. Thus, in "enum A {} n, m;",
4006 // "} n, m;" will end up in one unwrapped line.
4007}
4008
4009bool UnwrappedLineParser::parseStructLike() {
4010 // parseRecord falls through and does not yet add an unwrapped line as a
4011 // record declaration or definition can start a structural element.
4012 parseRecord();
4013 // This does not apply to Java, JavaScript and C#.
4014 if (Style.isJava() || Style.isJavaScript() || Style.isCSharp()) {
4015 if (FormatTok->is(tok::semi))
4016 nextToken();
4017 addUnwrappedLine();
4018 return true;
4019 }
4020 return false;
4021}
4022
4023namespace {
4024// A class used to set and restore the Token position when peeking
4025// ahead in the token source.
4026class ScopedTokenPosition {
4027 unsigned StoredPosition;
4028 FormatTokenSource *Tokens;
4029
4030public:
4031 ScopedTokenPosition(FormatTokenSource *Tokens) : Tokens(Tokens) {
4032 assert(Tokens && "Tokens expected to not be null");
4033 StoredPosition = Tokens->getPosition();
4034 }
4035
4036 ~ScopedTokenPosition() { Tokens->setPosition(StoredPosition); }
4037};
4038} // namespace
4039
4040// Look to see if we have [[ by looking ahead, if
4041// its not then rewind to the original position.
4042bool UnwrappedLineParser::tryToParseSimpleAttribute() {
4043 ScopedTokenPosition AutoPosition(Tokens);
4044 FormatToken *Tok = Tokens->getNextToken();
4045 // We already read the first [ check for the second.
4046 if (Tok->isNot(tok::l_square))
4047 return false;
4048 // Double check that the attribute is just something
4049 // fairly simple.
4050 while (Tok->isNot(tok::eof)) {
4051 if (Tok->is(tok::r_square))
4052 break;
4053 Tok = Tokens->getNextToken();
4054 }
4055 if (Tok->is(tok::eof))
4056 return false;
4057 Tok = Tokens->getNextToken();
4058 if (Tok->isNot(tok::r_square))
4059 return false;
4060 Tok = Tokens->getNextToken();
4061 if (Tok->is(tok::semi))
4062 return false;
4063 return true;
4064}
4065
4066void UnwrappedLineParser::parseJavaEnumBody() {
4067 assert(FormatTok->is(tok::l_brace));
4068 const FormatToken *OpeningBrace = FormatTok;
4069
4070 // Determine whether the enum is simple, i.e. does not have a semicolon or
4071 // constants with class bodies. Simple enums can be formatted like braced
4072 // lists, contracted to a single line, etc.
4073 unsigned StoredPosition = Tokens->getPosition();
4074 bool IsSimple = true;
4075 FormatToken *Tok = Tokens->getNextToken();
4076 while (Tok->isNot(tok::eof)) {
4077 if (Tok->is(tok::r_brace))
4078 break;
4079 if (Tok->isOneOf(tok::l_brace, tok::semi)) {
4080 IsSimple = false;
4081 break;
4082 }
4083 // FIXME: This will also mark enums with braces in the arguments to enum
4084 // constants as "not simple". This is probably fine in practice, though.
4085 Tok = Tokens->getNextToken();
4086 }
4087 FormatTok = Tokens->setPosition(StoredPosition);
4088
4089 if (IsSimple) {
4090 nextToken();
4091 parseBracedList();
4092 addUnwrappedLine();
4093 return;
4094 }
4095
4096 // Parse the body of a more complex enum.
4097 // First add a line for everything up to the "{".
4098 nextToken();
4099 addUnwrappedLine();
4100 ++Line->Level;
4101
4102 // Parse the enum constants.
4103 while (!eof()) {
4104 if (FormatTok->is(tok::l_brace)) {
4105 // Parse the constant's class body.
4106 parseBlock(/*MustBeDeclaration=*/true, /*AddLevels=*/1u,
4107 /*MunchSemi=*/false);
4108 } else if (FormatTok->is(tok::l_paren)) {
4109 parseParens();
4110 } else if (FormatTok->is(tok::comma)) {
4111 nextToken();
4112 addUnwrappedLine();
4113 } else if (FormatTok->is(tok::semi)) {
4114 nextToken();
4115 addUnwrappedLine();
4116 break;
4117 } else if (FormatTok->is(tok::r_brace)) {
4118 addUnwrappedLine();
4119 break;
4120 } else {
4121 nextToken();
4122 }
4123 }
4124
4125 // Parse the class body after the enum's ";" if any.
4126 parseLevel(OpeningBrace);
4127 nextToken();
4128 --Line->Level;
4129 addUnwrappedLine();
4130}
4131
4132void UnwrappedLineParser::parseRecord(bool ParseAsExpr, bool IsJavaRecord) {
4133 assert(!IsJavaRecord || FormatTok->is(Keywords.kw_record));
4134 const FormatToken &InitialToken = *FormatTok;
4135 nextToken();
4136
4137 FormatToken *ClassName =
4138 IsJavaRecord && FormatTok->is(tok::identifier) ? FormatTok : nullptr;
4139 bool IsDerived = false;
4140 auto IsNonMacroIdentifier = [](const FormatToken *Tok) {
4141 return Tok->is(tok::identifier) && Tok->TokenText != Tok->TokenText.upper();
4142 };
4143 // JavaScript/TypeScript supports anonymous classes like:
4144 // a = class extends foo { }
4145 bool JSPastExtendsOrImplements = false;
4146 // The actual identifier can be a nested name specifier, and in macros
4147 // it is often token-pasted.
4148 // An [[attribute]] can be before the identifier.
4149 while (FormatTok->isOneOf(tok::identifier, tok::coloncolon, tok::hashhash,
4150 tok::kw_alignas, tok::l_square) ||
4151 FormatTok->isAttribute() ||
4152 ((Style.isJava() || Style.isJavaScript()) &&
4153 FormatTok->isOneOf(tok::period, tok::comma))) {
4154 if (Style.isJavaScript() &&
4155 FormatTok->isOneOf(Keywords.kw_extends, Keywords.kw_implements)) {
4156 JSPastExtendsOrImplements = true;
4157 // JavaScript/TypeScript supports inline object types in
4158 // extends/implements positions:
4159 // class Foo implements {bar: number} { }
4160 nextToken();
4161 if (FormatTok->is(tok::l_brace)) {
4162 tryToParseBracedList();
4163 continue;
4164 }
4165 }
4166 if (FormatTok->is(tok::l_square) && handleCppAttributes())
4167 continue;
4168 auto *Previous = FormatTok;
4169 nextToken();
4170 switch (FormatTok->Tok.getKind()) {
4171 case tok::l_paren:
4172 // We can have macros in between 'class' and the class name.
4173 if (IsJavaRecord || !IsNonMacroIdentifier(Previous) ||
4174 // e.g. `struct macro(a) S { int i; };`
4175 Previous->Previous == &InitialToken) {
4176 parseParens();
4177 }
4178 break;
4179 case tok::coloncolon:
4180 case tok::hashhash:
4181 break;
4182 default:
4183 if (JSPastExtendsOrImplements || ClassName ||
4184 Previous->isNot(tok::identifier) || Previous->is(TT_AttributeMacro)) {
4185 break;
4186 }
4187 if (const auto Text = Previous->TokenText;
4188 Text.size() == 1 || Text != Text.upper()) {
4189 ClassName = Previous;
4190 }
4191 }
4192 }
4193
4194 auto IsListInitialization = [&] {
4195 if (!ClassName || IsDerived || JSPastExtendsOrImplements)
4196 return false;
4197 assert(FormatTok->is(tok::l_brace));
4198 const auto *Prev = FormatTok->getPreviousNonComment();
4199 assert(Prev);
4200 return Prev != ClassName && Prev->is(tok::identifier) &&
4201 Prev->isNot(Keywords.kw_final) && tryToParseBracedList();
4202 };
4203
4204 if (FormatTok->isOneOf(tok::colon, tok::less)) {
4205 int AngleNestingLevel = 0;
4206 do {
4207 if (FormatTok->is(tok::less))
4208 ++AngleNestingLevel;
4209 else if (FormatTok->is(tok::greater))
4210 --AngleNestingLevel;
4211
4212 if (AngleNestingLevel == 0) {
4213 if (FormatTok->is(tok::colon)) {
4214 IsDerived = true;
4215 } else if (!IsDerived && FormatTok->is(tok::identifier) &&
4216 FormatTok->Previous->is(tok::coloncolon)) {
4217 ClassName = FormatTok;
4218 } else if (FormatTok->is(tok::l_paren) &&
4219 IsNonMacroIdentifier(FormatTok->Previous)) {
4220 break;
4221 }
4222 }
4223 if (FormatTok->is(tok::l_brace)) {
4224 if (AngleNestingLevel == 0 && IsListInitialization())
4225 return;
4226 calculateBraceTypes(/*ExpectClassBody=*/true);
4227 if (!tryToParseBracedList())
4228 break;
4229 }
4230 if (FormatTok->is(tok::l_square)) {
4231 FormatToken *Previous = FormatTok->Previous;
4232 if (!Previous || (Previous->isNot(tok::r_paren) &&
4233 !Previous->isTypeOrIdentifier(LangOpts))) {
4234 // Don't try parsing a lambda if we had a closing parenthesis before,
4235 // it was probably a pointer to an array: int (*)[].
4236 if (!tryToParseLambda())
4237 continue;
4238 } else {
4239 parseSquare();
4240 continue;
4241 }
4242 }
4243 if (FormatTok->is(tok::semi))
4244 return;
4245 if (Style.isCSharp() && FormatTok->is(Keywords.kw_where)) {
4246 addUnwrappedLine();
4247 nextToken();
4248 parseCSharpGenericTypeConstraint();
4249 break;
4250 }
4251 nextToken();
4252 } while (!eof());
4253 }
4254
4255 auto GetBraceTypes =
4256 [](const FormatToken &RecordTok) -> std::pair<TokenType, TokenType> {
4257 switch (RecordTok.Tok.getKind()) {
4258 case tok::kw_class:
4259 return {TT_ClassLBrace, TT_ClassRBrace};
4260 case tok::kw_struct:
4261 return {TT_StructLBrace, TT_StructRBrace};
4262 case tok::kw_union:
4263 return {TT_UnionLBrace, TT_UnionRBrace};
4264 default:
4265 // Useful for e.g. interface.
4266 return {TT_RecordLBrace, TT_RecordRBrace};
4267 }
4268 };
4269 if (FormatTok->is(tok::l_brace)) {
4270 if (IsListInitialization())
4271 return;
4272 if (ClassName)
4273 ClassName->setFinalizedType(TT_ClassHeadName);
4274 auto [OpenBraceType, ClosingBraceType] = GetBraceTypes(InitialToken);
4275 FormatTok->setFinalizedType(OpenBraceType);
4276 if (ParseAsExpr) {
4277 parseChildBlock();
4278 } else {
4279 if (ShouldBreakBeforeBrace(Style, InitialToken,
4280 Tokens->peekNextToken()->is(tok::r_brace),
4281 IsJavaRecord)) {
4282 addUnwrappedLine();
4283 }
4284
4285 unsigned AddLevels = Style.IndentAccessModifiers ? 2u : 1u;
4286 parseBlock(/*MustBeDeclaration=*/true, AddLevels, /*MunchSemi=*/false);
4287 }
4288 setPreviousRBraceType(ClosingBraceType);
4289 }
4290 // There is no addUnwrappedLine() here so that we fall through to parsing a
4291 // structural element afterwards. Thus, in "class A {} n, m;",
4292 // "} n, m;" will end up in one unwrapped line.
4293}
4294
4295void UnwrappedLineParser::parseObjCMethod() {
4296 assert(FormatTok->isOneOf(tok::l_paren, tok::identifier) &&
4297 "'(' or identifier expected.");
4298 do {
4299 if (FormatTok->is(tok::semi)) {
4300 nextToken();
4301 addUnwrappedLine();
4302 return;
4303 } else if (FormatTok->is(tok::l_brace)) {
4304 if (Style.BraceWrapping.AfterFunction)
4305 addUnwrappedLine();
4306 parseBlock();
4307 addUnwrappedLine();
4308 return;
4309 } else {
4310 nextToken();
4311 }
4312 } while (!eof());
4313}
4314
4315void UnwrappedLineParser::parseObjCProtocolList() {
4316 assert(FormatTok->is(tok::less) && "'<' expected.");
4317 do {
4318 nextToken();
4319 // Early exit in case someone forgot a close angle.
4320 if (FormatTok->isOneOf(tok::semi, tok::l_brace, tok::objc_end))
4321 return;
4322 } while (!eof() && FormatTok->isNot(tok::greater));
4323 nextToken(); // Skip '>'.
4324}
4325
4326void UnwrappedLineParser::parseObjCUntilAtEnd() {
4327 do {
4328 if (FormatTok->is(tok::objc_end)) {
4329 nextToken();
4330 addUnwrappedLine();
4331 break;
4332 }
4333 if (FormatTok->is(tok::l_brace)) {
4334 parseBlock();
4335 // In ObjC interfaces, nothing should be following the "}".
4336 addUnwrappedLine();
4337 } else if (FormatTok->is(tok::r_brace)) {
4338 // Ignore stray "}". parseStructuralElement doesn't consume them.
4339 nextToken();
4340 addUnwrappedLine();
4341 } else if (FormatTok->isOneOf(tok::minus, tok::plus)) {
4342 nextToken();
4343 if (FormatTok->isOneOf(tok::l_paren, tok::identifier))
4344 parseObjCMethod();
4345 } else {
4346 parseStructuralElement();
4347 }
4348 } while (!eof());
4349}
4350
4351void UnwrappedLineParser::parseObjCInterfaceOrImplementation() {
4352 assert(FormatTok->isOneOf(tok::objc_interface, tok::objc_implementation));
4353 nextToken();
4354 nextToken(); // interface name
4355
4356 // @interface can be followed by a lightweight generic
4357 // specialization list, then either a base class or a category.
4358 if (FormatTok->is(tok::less))
4359 parseObjCLightweightGenerics();
4360 if (FormatTok->is(tok::colon)) {
4361 nextToken();
4362 nextToken(); // base class name
4363 // The base class can also have lightweight generics applied to it.
4364 if (FormatTok->is(tok::less))
4365 parseObjCLightweightGenerics();
4366 } else if (FormatTok->is(tok::l_paren)) {
4367 // Skip category, if present.
4368 parseParens();
4369 }
4370
4371 if (FormatTok->is(tok::less))
4372 parseObjCProtocolList();
4373
4374 if (FormatTok->is(tok::l_brace)) {
4375 if (Style.BraceWrapping.AfterObjCDeclaration)
4376 addUnwrappedLine();
4377 parseBlock(/*MustBeDeclaration=*/true);
4378 }
4379
4380 // With instance variables, this puts '}' on its own line. Without instance
4381 // variables, this ends the @interface line.
4382 addUnwrappedLine();
4383
4384 parseObjCUntilAtEnd();
4385}
4386
4387void UnwrappedLineParser::parseObjCLightweightGenerics() {
4388 assert(FormatTok->is(tok::less));
4389 // Unlike protocol lists, generic parameterizations support
4390 // nested angles:
4391 //
4392 // @interface Foo<ValueType : id <NSCopying, NSSecureCoding>> :
4393 // NSObject <NSCopying, NSSecureCoding>
4394 //
4395 // so we need to count how many open angles we have left.
4396 unsigned NumOpenAngles = 1;
4397 do {
4398 nextToken();
4399 // Early exit in case someone forgot a close angle.
4400 if (FormatTok->isOneOf(tok::semi, tok::l_brace, tok::objc_end))
4401 break;
4402 if (FormatTok->is(tok::less)) {
4403 ++NumOpenAngles;
4404 } else if (FormatTok->is(tok::greater)) {
4405 assert(NumOpenAngles > 0 && "'>' makes NumOpenAngles negative");
4406 --NumOpenAngles;
4407 }
4408 } while (!eof() && NumOpenAngles != 0);
4409 nextToken(); // Skip '>'.
4410}
4411
4412// Returns true for the declaration/definition form of @protocol,
4413// false for the expression form.
4414bool UnwrappedLineParser::parseObjCProtocol() {
4415 assert(FormatTok->is(tok::objc_protocol));
4416 nextToken();
4417
4418 if (FormatTok->is(tok::l_paren)) {
4419 // The expression form of @protocol, e.g. "Protocol* p = @protocol(foo);".
4420 return false;
4421 }
4422
4423 // The definition/declaration form,
4424 // @protocol Foo
4425 // - (int)someMethod;
4426 // @end
4427
4428 nextToken(); // protocol name
4429
4430 if (FormatTok->is(tok::less))
4431 parseObjCProtocolList();
4432
4433 // Check for protocol declaration.
4434 if (FormatTok->is(tok::semi)) {
4435 nextToken();
4436 addUnwrappedLine();
4437 return true;
4438 }
4439
4440 addUnwrappedLine();
4441 parseObjCUntilAtEnd();
4442 return true;
4443}
4444
4445void UnwrappedLineParser::parseJavaScriptEs6ImportExport() {
4446 bool IsImport = FormatTok->is(Keywords.kw_import);
4447 assert(IsImport || FormatTok->is(tok::kw_export));
4448 nextToken();
4449
4450 // Consume the "default" in "export default class/function".
4451 if (FormatTok->is(tok::kw_default))
4452 nextToken();
4453
4454 // Consume "async function", "function" and "default function", so that these
4455 // get parsed as free-standing JS functions, i.e. do not require a trailing
4456 // semicolon.
4457 if (FormatTok->is(Keywords.kw_async))
4458 nextToken();
4459 if (FormatTok->is(Keywords.kw_function)) {
4460 nextToken();
4461 return;
4462 }
4463
4464 // For imports, `export *`, `export {...}`, consume the rest of the line up
4465 // to the terminating `;`. For everything else, just return and continue
4466 // parsing the structural element, i.e. the declaration or expression for
4467 // `export default`.
4468 if (!IsImport && FormatTok->isNoneOf(tok::l_brace, tok::star) &&
4469 !FormatTok->isStringLiteral() &&
4470 !(FormatTok->is(Keywords.kw_type) &&
4471 Tokens->peekNextToken()->isOneOf(tok::l_brace, tok::star))) {
4472 return;
4473 }
4474
4475 while (!eof()) {
4476 if (FormatTok->is(tok::semi))
4477 return;
4478 if (Line->Tokens.empty()) {
4479 // Common issue: Automatic Semicolon Insertion wrapped the line, so the
4480 // import statement should terminate.
4481 return;
4482 }
4483 if (FormatTok->is(tok::l_brace)) {
4484 FormatTok->setBlockKind(BK_Block);
4485 nextToken();
4486 parseBracedList();
4487 } else {
4488 nextToken();
4489 }
4490 }
4491}
4492
4493void UnwrappedLineParser::parseStatementMacro() {
4494 nextToken();
4495 if (FormatTok->is(tok::l_paren))
4496 parseParens();
4497 if (FormatTok->is(tok::semi))
4498 nextToken();
4499 addUnwrappedLine();
4500}
4501
4502void UnwrappedLineParser::parseVerilogHierarchyIdentifier() {
4503 // consume things like a::`b.c[d:e] or a::*
4504 while (true) {
4505 if (FormatTok->isOneOf(tok::star, tok::period, tok::periodstar,
4506 tok::coloncolon, tok::hash) ||
4507 Keywords.isVerilogIdentifier(*FormatTok)) {
4508 nextToken();
4509 } else if (FormatTok->is(tok::l_square)) {
4510 parseSquare();
4511 } else {
4512 break;
4513 }
4514 }
4515}
4516
4517void UnwrappedLineParser::parseVerilogSensitivityList() {
4518 if (FormatTok->isNot(tok::at))
4519 return;
4520 nextToken();
4521 // A block event expression has 2 at signs.
4522 if (FormatTok->is(tok::at))
4523 nextToken();
4524 switch (FormatTok->Tok.getKind()) {
4525 case tok::star:
4526 nextToken();
4527 break;
4528 case tok::l_paren:
4529 parseParens();
4530 break;
4531 default:
4532 parseVerilogHierarchyIdentifier();
4533 break;
4534 }
4535}
4536
4537unsigned UnwrappedLineParser::parseVerilogHierarchyHeader() {
4538 unsigned AddLevels = 0;
4539
4540 if (FormatTok->is(Keywords.kw_clocking)) {
4541 nextToken();
4542 if (Keywords.isVerilogIdentifier(*FormatTok))
4543 nextToken();
4544 parseVerilogSensitivityList();
4545 if (FormatTok->is(tok::semi))
4546 nextToken();
4547 } else if (FormatTok->isOneOf(tok::kw_case, Keywords.kw_casex,
4548 Keywords.kw_casez, Keywords.kw_randcase,
4549 Keywords.kw_randsequence)) {
4550 if (Style.IndentCaseLabels)
4551 AddLevels++;
4552 nextToken();
4553 if (FormatTok->is(tok::l_paren)) {
4554 FormatTok->setFinalizedType(TT_ConditionLParen);
4555 parseParens();
4556 }
4557 if (FormatTok->isOneOf(Keywords.kw_inside, Keywords.kw_matches))
4558 nextToken();
4559 // The case header has no semicolon.
4560 } else {
4561 // "module" etc.
4562 nextToken();
4563 // all the words like the name of the module and specifiers like
4564 // "automatic" and the width of function return type
4565 while (true) {
4566 if (FormatTok->is(tok::l_square)) {
4567 auto Prev = FormatTok->getPreviousNonComment();
4568 if (Prev && Keywords.isVerilogIdentifier(*Prev))
4569 Prev->setFinalizedType(TT_VerilogDimensionedTypeName);
4570 parseSquare();
4571 } else if (Keywords.isVerilogIdentifier(*FormatTok) ||
4572 FormatTok->isOneOf(tok::hash, tok::hashhash, tok::coloncolon,
4573 Keywords.kw_automatic, tok::kw_static)) {
4574 nextToken();
4575 } else {
4576 break;
4577 }
4578 }
4579
4580 auto NewLine = [this]() {
4581 addUnwrappedLine();
4582 Line->IsContinuation = true;
4583 };
4584
4585 // package imports
4586 while (FormatTok->is(Keywords.kw_import)) {
4587 NewLine();
4588 nextToken();
4589 parseVerilogHierarchyIdentifier();
4590 if (FormatTok->is(tok::semi))
4591 nextToken();
4592 }
4593
4594 // parameters and ports
4595 if (FormatTok->is(Keywords.kw_verilogHash)) {
4596 NewLine();
4597 nextToken();
4598 if (FormatTok->is(tok::l_paren)) {
4599 FormatTok->setFinalizedType(TT_VerilogMultiLineListLParen);
4600 parseParens();
4601 }
4602 }
4603 if (FormatTok->is(tok::l_paren)) {
4604 NewLine();
4605 FormatTok->setFinalizedType(TT_VerilogMultiLineListLParen);
4606 parseParens();
4607 }
4608
4609 // extends and implements
4610 if (FormatTok->is(Keywords.kw_extends)) {
4611 NewLine();
4612 nextToken();
4613 parseVerilogHierarchyIdentifier();
4614 if (FormatTok->is(tok::l_paren))
4615 parseParens();
4616 }
4617 if (FormatTok->is(Keywords.kw_implements)) {
4618 NewLine();
4619 do {
4620 nextToken();
4621 parseVerilogHierarchyIdentifier();
4622 } while (FormatTok->is(tok::comma));
4623 }
4624
4625 // Coverage event for cover groups.
4626 if (FormatTok->is(tok::at)) {
4627 NewLine();
4628 parseVerilogSensitivityList();
4629 }
4630
4631 if (FormatTok->is(tok::semi))
4632 nextToken(/*LevelDifference=*/1);
4633 addUnwrappedLine();
4634 }
4635
4636 return AddLevels;
4637}
4638
4639void UnwrappedLineParser::parseVerilogTable() {
4640 assert(FormatTok->is(Keywords.kw_table));
4641 nextToken(/*LevelDifference=*/1);
4642 addUnwrappedLine();
4643
4644 auto InitialLevel = Line->Level++;
4645 while (!eof() && !Keywords.isVerilogEnd(*FormatTok)) {
4646 FormatToken *Tok = FormatTok;
4647 nextToken();
4648 if (Tok->is(tok::semi))
4649 addUnwrappedLine();
4650 else if (Tok->isOneOf(tok::star, tok::colon, tok::question, tok::minus))
4651 Tok->setFinalizedType(TT_VerilogTableItem);
4652 }
4653 Line->Level = InitialLevel;
4654 nextToken(/*LevelDifference=*/-1);
4655 addUnwrappedLine();
4656}
4657
4658void UnwrappedLineParser::parseVerilogCaseLabel() {
4659 // The label will get unindented in AnnotatingParser. If there are no leading
4660 // spaces, indent the rest here so that things inside the block will be
4661 // indented relative to things outside. We don't use parseLabel because we
4662 // don't know whether this colon is a label or a ternary expression at this
4663 // point.
4664 auto OrigLevel = Line->Level;
4665 auto FirstLine = CurrentLines->size();
4666 if (Line->Level == 0 || (Line->InPPDirective && Line->Level <= 1))
4667 ++Line->Level;
4668 else if (!Style.IndentCaseBlocks && Keywords.isVerilogBegin(*FormatTok))
4669 --Line->Level;
4670 parseStructuralElement();
4671 // Restore the indentation in both the new line and the line that has the
4672 // label.
4673 if (CurrentLines->size() > FirstLine)
4674 (*CurrentLines)[FirstLine].Level = OrigLevel;
4675 Line->Level = OrigLevel;
4676}
4677
4678void UnwrappedLineParser::parseVerilogExtern() {
4679 assert(
4680 FormatTok->isOneOf(tok::kw_extern, tok::kw_export, Keywords.kw_import));
4681 nextToken();
4682 // "DPI-C"
4683 if (FormatTok->is(tok::string_literal))
4684 nextToken();
4685 skipVerilogQualifiers();
4686 if (Keywords.isVerilogIdentifier(*FormatTok))
4687 nextToken();
4688 if (FormatTok->is(tok::equal))
4689 nextToken();
4690 if (Keywords.isVerilogHierarchy(*FormatTok))
4691 parseVerilogHierarchyHeader();
4692}
4693
4694void UnwrappedLineParser::skipVerilogQualifiers() {
4695 while (FormatTok->isOneOf(tok::kw_protected, tok::kw_virtual, tok::kw_static,
4696 Keywords.kw_rand, Keywords.kw_context,
4697 Keywords.kw_pure, Keywords.kw_randc,
4698 Keywords.kw_local)) {
4699 nextToken();
4700 }
4701}
4702
4703bool UnwrappedLineParser::containsExpansion(const UnwrappedLine &Line) const {
4704 for (const auto &N : Line.Tokens) {
4705 if (N.Tok->MacroCtx)
4706 return true;
4707 for (const UnwrappedLine &Child : N.Children)
4708 if (containsExpansion(Child))
4709 return true;
4710 }
4711 return false;
4712}
4713
4714void UnwrappedLineParser::addUnwrappedLine(LineLevel AdjustLevel) {
4715 if (Line->Tokens.empty())
4716 return;
4717 LLVM_DEBUG({
4718 if (!parsingPPDirective()) {
4719 llvm::dbgs() << "Adding unwrapped line:\n";
4720 printDebugInfo(*Line);
4721 }
4722 });
4723
4724 // If this line closes a block when in Whitesmiths mode, remember that
4725 // information so that the level can be decreased after the line is added.
4726 // This has to happen after the addition of the line since the line itself
4727 // needs to be indented.
4728 bool ClosesWhitesmithsBlock =
4729 Line->MatchingOpeningBlockLineIndex != UnwrappedLine::kInvalidIndex &&
4730 Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths;
4731
4732 // If the current line was expanded from a macro call, we use it to
4733 // reconstruct an unwrapped line from the structure of the expanded unwrapped
4734 // line and the unexpanded token stream.
4735 if (!parsingPPDirective() && !InExpansion && containsExpansion(*Line)) {
4736 if (!Reconstruct)
4737 Reconstruct.emplace(Line->Level, Unexpanded);
4738 Reconstruct->addLine(*Line);
4739
4740 // While the reconstructed unexpanded lines are stored in the normal
4741 // flow of lines, the expanded lines are stored on the side to be analyzed
4742 // in an extra step.
4743 CurrentExpandedLines.push_back(std::move(*Line));
4744
4745 if (Reconstruct->finished()) {
4746 UnwrappedLine Reconstructed = std::move(*Reconstruct).takeResult();
4747 assert(!Reconstructed.Tokens.empty() &&
4748 "Reconstructed must at least contain the macro identifier.");
4749 assert(!parsingPPDirective());
4750 LLVM_DEBUG({
4751 llvm::dbgs() << "Adding unexpanded line:\n";
4752 printDebugInfo(Reconstructed);
4753 });
4754 ExpandedLines[Reconstructed.Tokens.begin()->Tok] = CurrentExpandedLines;
4755 Lines.push_back(std::move(Reconstructed));
4756 CurrentExpandedLines.clear();
4757 Reconstruct.reset();
4758 }
4759 } else {
4760 // At the top level we only get here when no unexpansion is going on, or
4761 // when conditional formatting led to unfinished macro reconstructions.
4762 assert(!Reconstruct || (CurrentLines != &Lines) || !PPStack.empty());
4763 CurrentLines->push_back(std::move(*Line));
4764 }
4765 Line->Tokens.clear();
4766 Line->MatchingOpeningBlockLineIndex = UnwrappedLine::kInvalidIndex;
4767 Line->FirstStartColumn = 0;
4768 Line->IsContinuation = false;
4769 Line->SeenDecltypeAuto = false;
4770 Line->IsModuleOrImportDecl = false;
4771
4772 if (ClosesWhitesmithsBlock && AdjustLevel == LineLevel::Remove)
4773 --Line->Level;
4774 if (!parsingPPDirective() && !PreprocessorDirectives.empty()) {
4775 CurrentLines->append(
4776 std::make_move_iterator(PreprocessorDirectives.begin()),
4777 std::make_move_iterator(PreprocessorDirectives.end()));
4778 PreprocessorDirectives.clear();
4779 }
4780 // Disconnect the current token from the last token on the previous line.
4781 FormatTok->Previous = nullptr;
4782}
4783
4784bool UnwrappedLineParser::eof() const { return FormatTok->is(tok::eof); }
4785
4786bool UnwrappedLineParser::isOnNewLine(const FormatToken &FormatTok) {
4787 return (Line->InPPDirective || FormatTok.HasUnescapedNewline) &&
4788 FormatTok.NewlinesBefore > 0;
4789}
4790
4791// Checks if \p FormatTok is a line comment that continues the line comment
4792// section on \p Line.
4793static bool
4795 const UnwrappedLine &Line, const FormatStyle &Style,
4796 const llvm::Regex &CommentPragmasRegex) {
4797 if (Line.Tokens.empty() || Style.ReflowComments != FormatStyle::RCS_Always)
4798 return false;
4799
4800 StringRef IndentContent = FormatTok.TokenText;
4801 if (FormatTok.TokenText.starts_with("//") ||
4802 FormatTok.TokenText.starts_with("/*")) {
4803 IndentContent = FormatTok.TokenText.substr(2);
4804 }
4805 if (CommentPragmasRegex.match(IndentContent))
4806 return false;
4807
4808 // If Line starts with a line comment, then FormatTok continues the comment
4809 // section if its original column is greater or equal to the original start
4810 // column of the line.
4811 //
4812 // Define the min column token of a line as follows: if a line ends in '{' or
4813 // contains a '{' followed by a line comment, then the min column token is
4814 // that '{'. Otherwise, the min column token of the line is the first token of
4815 // the line.
4816 //
4817 // If Line starts with a token other than a line comment, then FormatTok
4818 // continues the comment section if its original column is greater than the
4819 // original start column of the min column token of the line.
4820 //
4821 // For example, the second line comment continues the first in these cases:
4822 //
4823 // // first line
4824 // // second line
4825 //
4826 // and:
4827 //
4828 // // first line
4829 // // second line
4830 //
4831 // and:
4832 //
4833 // int i; // first line
4834 // // second line
4835 //
4836 // and:
4837 //
4838 // do { // first line
4839 // // second line
4840 // int i;
4841 // } while (true);
4842 //
4843 // and:
4844 //
4845 // enum {
4846 // a, // first line
4847 // // second line
4848 // b
4849 // };
4850 //
4851 // The second line comment doesn't continue the first in these cases:
4852 //
4853 // // first line
4854 // // second line
4855 //
4856 // and:
4857 //
4858 // int i; // first line
4859 // // second line
4860 //
4861 // and:
4862 //
4863 // do { // first line
4864 // // second line
4865 // int i;
4866 // } while (true);
4867 //
4868 // and:
4869 //
4870 // enum {
4871 // a, // first line
4872 // // second line
4873 // };
4874 const FormatToken *MinColumnToken = Line.Tokens.front().Tok;
4875
4876 // Scan for '{//'. If found, use the column of '{' as a min column for line
4877 // comment section continuation.
4878 const FormatToken *PreviousToken = nullptr;
4879 for (const UnwrappedLineNode &Node : Line.Tokens) {
4880 if (PreviousToken && PreviousToken->is(tok::l_brace) &&
4881 isLineComment(*Node.Tok)) {
4882 MinColumnToken = PreviousToken;
4883 break;
4884 }
4885 PreviousToken = Node.Tok;
4886
4887 // Grab the last newline preceding a token in this unwrapped line.
4888 if (Node.Tok->NewlinesBefore > 0)
4889 MinColumnToken = Node.Tok;
4890 }
4891 if (PreviousToken && PreviousToken->is(tok::l_brace))
4892 MinColumnToken = PreviousToken;
4893
4894 return continuesLineComment(FormatTok, /*Previous=*/Line.Tokens.back().Tok,
4895 MinColumnToken);
4896}
4897
4898void UnwrappedLineParser::flushComments(bool NewlineBeforeNext) {
4899 bool JustComments = Line->Tokens.empty();
4900 for (FormatToken *Tok : CommentsBeforeNextToken) {
4901 // Line comments that belong to the same line comment section are put on the
4902 // same line since later we might want to reflow content between them.
4903 // Additional fine-grained breaking of line comment sections is controlled
4904 // by the class BreakableLineCommentSection in case it is desirable to keep
4905 // several line comment sections in the same unwrapped line.
4906 //
4907 // FIXME: Consider putting separate line comment sections as children to the
4908 // unwrapped line instead.
4909 Tok->ContinuesLineCommentSection =
4910 continuesLineCommentSection(*Tok, *Line, Style, CommentPragmasRegex);
4911 if (isOnNewLine(*Tok) && JustComments && !Tok->ContinuesLineCommentSection)
4912 addUnwrappedLine();
4913 pushToken(Tok);
4914 }
4915 if (NewlineBeforeNext && JustComments)
4916 addUnwrappedLine();
4917 CommentsBeforeNextToken.clear();
4918}
4919
4920void UnwrappedLineParser::nextToken(int LevelDifference) {
4921 if (eof())
4922 return;
4923 flushComments(isOnNewLine(*FormatTok));
4924 pushToken(FormatTok);
4925 FormatToken *Previous = FormatTok;
4926 if (!Style.isJavaScript())
4927 readToken(LevelDifference);
4928 else
4929 readTokenWithJavaScriptASI();
4930 FormatTok->Previous = Previous;
4931 if (Style.isVerilog()) {
4932 // Blocks in Verilog can have `begin` and `end` instead of braces. For
4933 // keywords like `begin`, we can't treat them the same as left braces
4934 // because some contexts require one of them. For example structs use
4935 // braces and if blocks use keywords, and a left brace can occur in an if
4936 // statement, but it is not a block. For keywords like `end`, we simply
4937 // treat them the same as right braces.
4938 if (Keywords.isVerilogEnd(*FormatTok))
4939 FormatTok->Tok.setKind(tok::r_brace);
4940 }
4941}
4942
4943void UnwrappedLineParser::distributeComments(
4944 const ArrayRef<FormatToken *> &Comments, const FormatToken *NextTok) {
4945 // Whether or not a line comment token continues a line is controlled by
4946 // the method continuesLineCommentSection, with the following caveat:
4947 //
4948 // Define a trail of Comments to be a nonempty proper postfix of Comments such
4949 // that each comment line from the trail is aligned with the next token, if
4950 // the next token exists. If a trail exists, the beginning of the maximal
4951 // trail is marked as a start of a new comment section.
4952 //
4953 // For example in this code:
4954 //
4955 // int a; // line about a
4956 // // line 1 about b
4957 // // line 2 about b
4958 // int b;
4959 //
4960 // the two lines about b form a maximal trail, so there are two sections, the
4961 // first one consisting of the single comment "// line about a" and the
4962 // second one consisting of the next two comments.
4963 if (Comments.empty())
4964 return;
4965 bool ShouldPushCommentsInCurrentLine = true;
4966 bool HasTrailAlignedWithNextToken = false;
4967 unsigned StartOfTrailAlignedWithNextToken = 0;
4968 if (NextTok) {
4969 // We are skipping the first element intentionally.
4970 for (unsigned i = Comments.size() - 1; i > 0; --i) {
4971 if (Comments[i]->OriginalColumn == NextTok->OriginalColumn) {
4972 HasTrailAlignedWithNextToken = true;
4973 StartOfTrailAlignedWithNextToken = i;
4974 }
4975 }
4976 }
4977 for (unsigned i = 0, e = Comments.size(); i < e; ++i) {
4978 FormatToken *FormatTok = Comments[i];
4979 if (HasTrailAlignedWithNextToken && i == StartOfTrailAlignedWithNextToken) {
4980 FormatTok->ContinuesLineCommentSection = false;
4981 } else {
4982 FormatTok->ContinuesLineCommentSection = continuesLineCommentSection(
4983 *FormatTok, *Line, Style, CommentPragmasRegex);
4984 }
4985 if (!FormatTok->ContinuesLineCommentSection &&
4986 (isOnNewLine(*FormatTok) || FormatTok->IsFirst)) {
4987 ShouldPushCommentsInCurrentLine = false;
4988 }
4989 if (ShouldPushCommentsInCurrentLine)
4990 pushToken(FormatTok);
4991 else
4992 CommentsBeforeNextToken.push_back(FormatTok);
4993 }
4994}
4995
4996void UnwrappedLineParser::readToken(int LevelDifference) {
4998 bool PreviousWasComment = false;
4999 bool FirstNonCommentOnLine = false;
5000 do {
5001 FormatTok = Tokens->getNextToken();
5002 assert(FormatTok);
5003 while (FormatTok->isOneOf(TT_ConflictStart, TT_ConflictEnd,
5004 TT_ConflictAlternative)) {
5005 if (FormatTok->is(TT_ConflictStart))
5006 conditionalCompilationStart(/*Unreachable=*/false);
5007 else if (FormatTok->is(TT_ConflictAlternative))
5008 conditionalCompilationAlternative();
5009 else if (FormatTok->is(TT_ConflictEnd))
5010 conditionalCompilationEnd();
5011 FormatTok = Tokens->getNextToken();
5012 FormatTok->MustBreakBefore = true;
5013 FormatTok->MustBreakBeforeFinalized = true;
5014 }
5015
5016 auto IsFirstNonCommentOnLine = [](bool FirstNonCommentOnLine,
5017 const FormatToken &Tok,
5018 bool PreviousWasComment) {
5019 auto IsFirstOnLine = [](const FormatToken &Tok) {
5020 return Tok.HasUnescapedNewline || Tok.IsFirst;
5021 };
5022
5023 // Consider preprocessor directives preceded by block comments as first
5024 // on line.
5025 if (PreviousWasComment)
5026 return FirstNonCommentOnLine || IsFirstOnLine(Tok);
5027 return IsFirstOnLine(Tok);
5028 };
5029
5030 FirstNonCommentOnLine = IsFirstNonCommentOnLine(
5031 FirstNonCommentOnLine, *FormatTok, PreviousWasComment);
5032 PreviousWasComment = FormatTok->is(tok::comment);
5033
5034 while (!Line->InPPDirective && FormatTok->is(tok::hash) &&
5035 FirstNonCommentOnLine) {
5036 // In Verilog, the backtick is used for macro invocations. In TableGen,
5037 // the single hash is used for the paste operator.
5038 const auto *Next = Tokens->peekNextToken();
5039 if ((Style.isVerilog() && !Keywords.isVerilogPPDirective(*Next)) ||
5040 (Style.isTableGen() &&
5041 Next->isNoneOf(tok::kw_else, tok::pp_define, tok::pp_ifdef,
5042 tok::pp_ifndef, tok::pp_endif))) {
5043 break;
5044 }
5045 distributeComments(Comments, FormatTok);
5046 Comments.clear();
5047 // If there is an unfinished unwrapped line, we flush the preprocessor
5048 // directives only after that unwrapped line was finished later.
5049 bool SwitchToPreprocessorLines = !Line->Tokens.empty();
5050 ScopedLineState BlockState(*this, SwitchToPreprocessorLines);
5051 assert((LevelDifference >= 0 ||
5052 static_cast<unsigned>(-LevelDifference) <= Line->Level) &&
5053 "LevelDifference makes Line->Level negative");
5054 Line->Level += LevelDifference;
5055 // Comments stored before the preprocessor directive need to be output
5056 // before the preprocessor directive, at the same level as the
5057 // preprocessor directive, as we consider them to apply to the directive.
5058 if (Style.IndentPPDirectives == FormatStyle::PPDIS_BeforeHash &&
5059 PPBranchLevel > 0) {
5060 Line->Level += PPBranchLevel;
5061 }
5062 assert(Line->Level >= Line->UnbracedBodyLevel);
5063 Line->Level -= Line->UnbracedBodyLevel;
5064 flushComments(isOnNewLine(*FormatTok));
5065 const bool IsEndIf = Tokens->peekNextToken()->is(tok::pp_endif);
5066 parsePPDirective();
5067 PreviousWasComment = FormatTok->is(tok::comment);
5068 FirstNonCommentOnLine = IsFirstNonCommentOnLine(
5069 FirstNonCommentOnLine, *FormatTok, PreviousWasComment);
5070 // If the #endif of a potential include guard is the last thing in the
5071 // file, then we found an include guard.
5072 if (IsEndIf && IncludeGuard == IG_Defined && PPBranchLevel == -1 &&
5073 getIncludeGuardState(Style.IndentPPDirectives) == IG_Inited &&
5074 (eof() ||
5075 (PreviousWasComment &&
5076 Tokens->peekNextToken(/*SkipComment=*/true)->is(tok::eof)))) {
5077 IncludeGuard = IG_Found;
5078 }
5079 }
5080
5081 if (!PPStack.empty() && (PPStack.back().Kind == PP_Unreachable) &&
5082 !Line->InPPDirective) {
5083 continue;
5084 }
5085
5086 if (FormatTok->is(tok::identifier) &&
5087 Macros.defined(FormatTok->TokenText) &&
5088 // FIXME: Allow expanding macros in preprocessor directives.
5089 !Line->InPPDirective) {
5090 FormatToken *ID = FormatTok;
5091 unsigned Position = Tokens->getPosition();
5092
5093 // To correctly parse the code, we need to replace the tokens of the macro
5094 // call with its expansion.
5095 auto PreCall = std::move(Line);
5096 Line.reset(new UnwrappedLine);
5097 bool OldInExpansion = InExpansion;
5098 InExpansion = true;
5099 // We parse the macro call into a new line.
5100 auto Args = parseMacroCall();
5101 InExpansion = OldInExpansion;
5102 assert(Line->Tokens.front().Tok == ID);
5103 // And remember the unexpanded macro call tokens.
5104 auto UnexpandedLine = std::move(Line);
5105 // Reset to the old line.
5106 Line = std::move(PreCall);
5107
5108 LLVM_DEBUG({
5109 llvm::dbgs() << "Macro call: " << ID->TokenText << "(";
5110 if (Args) {
5111 llvm::dbgs() << "(";
5112 for (const auto &Arg : Args.value())
5113 for (const auto &T : Arg)
5114 llvm::dbgs() << T->TokenText << " ";
5115 llvm::dbgs() << ")";
5116 }
5117 llvm::dbgs() << "\n";
5118 });
5119 if (Macros.objectLike(ID->TokenText) && Args &&
5120 !Macros.hasArity(ID->TokenText, Args->size())) {
5121 // The macro is either
5122 // - object-like, but we got argumnets, or
5123 // - overloaded to be both object-like and function-like, but none of
5124 // the function-like arities match the number of arguments.
5125 // Thus, expand as object-like macro.
5126 LLVM_DEBUG(llvm::dbgs()
5127 << "Macro \"" << ID->TokenText
5128 << "\" not overloaded for arity " << Args->size()
5129 << "or not function-like, using object-like overload.");
5130 Args.reset();
5131 UnexpandedLine->Tokens.resize(1);
5132 Tokens->setPosition(Position);
5133 nextToken();
5134 assert(!Args && Macros.objectLike(ID->TokenText));
5135 }
5136 if ((!Args && Macros.objectLike(ID->TokenText)) ||
5137 (Args && Macros.hasArity(ID->TokenText, Args->size()))) {
5138 // Next, we insert the expanded tokens in the token stream at the
5139 // current position, and continue parsing.
5140 Unexpanded[ID] = std::move(UnexpandedLine);
5142 Macros.expand(ID, std::move(Args));
5143 if (!Expansion.empty())
5144 FormatTok = Tokens->insertTokens(Expansion);
5145
5146 LLVM_DEBUG({
5147 llvm::dbgs() << "Expanded: ";
5148 for (const auto &T : Expansion)
5149 llvm::dbgs() << T->TokenText << " ";
5150 llvm::dbgs() << "\n";
5151 });
5152 } else {
5153 LLVM_DEBUG({
5154 llvm::dbgs() << "Did not expand macro \"" << ID->TokenText
5155 << "\", because it was used ";
5156 if (Args)
5157 llvm::dbgs() << "with " << Args->size();
5158 else
5159 llvm::dbgs() << "without";
5160 llvm::dbgs() << " arguments, which doesn't match any definition.\n";
5161 });
5162 Tokens->setPosition(Position);
5163 FormatTok = ID;
5164 }
5165 }
5166
5167 if (FormatTok->isNot(tok::comment)) {
5168 distributeComments(Comments, FormatTok);
5169 Comments.clear();
5170 return;
5171 }
5172
5173 Comments.push_back(FormatTok);
5174 } while (!eof());
5175
5176 distributeComments(Comments, nullptr);
5177 Comments.clear();
5178}
5179
5180namespace {
5181template <typename Iterator>
5182void pushTokens(Iterator Begin, Iterator End,
5184 for (auto I = Begin; I != End; ++I) {
5185 Into.push_back(I->Tok);
5186 for (const auto &Child : I->Children)
5187 pushTokens(Child.Tokens.begin(), Child.Tokens.end(), Into);
5188 }
5189}
5190} // namespace
5191
5192std::optional<llvm::SmallVector<llvm::SmallVector<FormatToken *, 8>, 1>>
5193UnwrappedLineParser::parseMacroCall() {
5194 std::optional<llvm::SmallVector<llvm::SmallVector<FormatToken *, 8>, 1>> Args;
5195 assert(Line->Tokens.empty());
5196 nextToken();
5197 if (FormatTok->isNot(tok::l_paren))
5198 return Args;
5199 unsigned Position = Tokens->getPosition();
5200 FormatToken *Tok = FormatTok;
5201 nextToken();
5202 Args.emplace();
5203 auto ArgStart = std::prev(Line->Tokens.end());
5204
5205 int Parens = 0;
5206 do {
5207 switch (FormatTok->Tok.getKind()) {
5208 case tok::l_paren:
5209 ++Parens;
5210 nextToken();
5211 break;
5212 case tok::r_paren: {
5213 if (Parens > 0) {
5214 --Parens;
5215 nextToken();
5216 break;
5217 }
5218 Args->push_back({});
5219 pushTokens(std::next(ArgStart), Line->Tokens.end(), Args->back());
5220 nextToken();
5221 return Args;
5222 }
5223 case tok::comma: {
5224 if (Parens > 0) {
5225 nextToken();
5226 break;
5227 }
5228 Args->push_back({});
5229 pushTokens(std::next(ArgStart), Line->Tokens.end(), Args->back());
5230 nextToken();
5231 ArgStart = std::prev(Line->Tokens.end());
5232 break;
5233 }
5234 default:
5235 nextToken();
5236 break;
5237 }
5238 } while (!eof());
5239 Line->Tokens.resize(1);
5240 Tokens->setPosition(Position);
5241 FormatTok = Tok;
5242 return {};
5243}
5244
5245void UnwrappedLineParser::pushToken(FormatToken *Tok) {
5246 Line->Tokens.push_back(UnwrappedLineNode(Tok));
5247 if (AtEndOfPPLine) {
5248 auto &Tok = *Line->Tokens.back().Tok;
5249 Tok.MustBreakBefore = true;
5250 Tok.MustBreakBeforeFinalized = true;
5251 Tok.FirstAfterPPLine = true;
5252 AtEndOfPPLine = false;
5253 }
5254}
5255
5256} // end namespace format
5257} // end namespace clang
This file defines the FormatTokenSource interface, which provides a token stream as well as the abili...
This file contains the declaration of the FormatToken, a wrapper around Token with additional informa...
FormatToken()
Token Tok
The Token.
unsigned OriginalColumn
The original 0-based column of this token, including expanded tabs.
FormatToken * Previous
The previous token in the unwrapped line.
FormatToken * Next
The next token in the unwrapped line.
This file contains the main building blocks of macro support in clang-format.
static bool HasAttribute(const QualType &T)
This file implements a token annotator, i.e.
Defines the clang::TokenKind enum and support functions.
This file contains the declaration of the UnwrappedLineParser, which turns a stream of tokens into Un...
Implements an efficient mapping from strings to IdentifierInfo nodes.
Parser - This implements a parser for the C family of languages.
Definition Parser.h:256
This class handles loading and caching of source files into memory.
Token - This structure provides full information about a lexed token.
Definition Token.h:36
IdentifierInfo * getIdentifierInfo() const
Definition Token.h:197
bool isLiteral() const
Return true if this is a "literal", like a numeric constant, string, etc.
Definition Token.h:126
bool is(tok::TokenKind K) const
is/isNot - Predicates to check if this token is a specific kind, as in "if (Tok.is(tok::l_brace)) {....
Definition Token.h:104
tok::TokenKind getKind() const
Definition Token.h:99
bool isOneOf(Ts... Ks) const
Definition Token.h:105
bool isNot(tok::TokenKind K) const
Definition Token.h:111
CompoundStatementIndenter(UnwrappedLineParser *Parser, const FormatStyle &Style, unsigned &LineLevel)
CompoundStatementIndenter(UnwrappedLineParser *Parser, unsigned &LineLevel, bool WrapBrace, bool IndentBrace)
ScopedLineState(UnwrappedLineParser &Parser, bool SwitchToPreprocessorLines=false)
Interface for users of the UnwrappedLineParser to receive the parsed lines.
UnwrappedLineParser(SourceManager &SourceMgr, const FormatStyle &Style, const AdditionalKeywords &Keywords, unsigned FirstStartColumn, ArrayRef< FormatToken * > Tokens, UnwrappedLineConsumer &Callback, llvm::SpecificBumpPtrAllocator< FormatToken > &Allocator, IdentifierTable &IdentTable)
static void hash_combine(std::size_t &seed, const T &v)
static bool mustBeJSIdentOrValue(const AdditionalKeywords &Keywords, const FormatToken *FormatTok)
std::ostream & operator<<(std::ostream &Stream, const UnwrappedLine &Line)
static bool tokenCanStartNewLine(const FormatToken &Tok)
static bool continuesLineCommentSection(const FormatToken &FormatTok, const UnwrappedLine &Line, const FormatStyle &Style, const llvm::Regex &CommentPragmasRegex)
static bool isC78Type(const FormatToken &Tok)
static bool isJSDeclOrStmt(const AdditionalKeywords &Keywords, const FormatToken *FormatTok)
LangOptions getFormattingLangOpts(const FormatStyle &Style=getLLVMStyle())
Returns the LangOpts that the formatter expects you to set.
Definition Format.cpp:4518
static void markOptionalBraces(FormatToken *LeftBrace)
static bool mustBeJSIdent(const AdditionalKeywords &Keywords, const FormatToken *FormatTok)
static bool isIIFE(const UnwrappedLine &Line, const AdditionalKeywords &Keywords)
static bool isC78ParameterDecl(const FormatToken *Tok, const FormatToken *Next, const FormatToken *FuncName)
static bool isGoogScope(const UnwrappedLine &Line)
static FormatToken * getLastNonComment(const UnwrappedLine &Line)
TokenType
Determines the semantic type of a syntactic token, e.g.
static bool ShouldBreakBeforeBrace(const FormatStyle &Style, const FormatToken &InitialToken, bool IsEmptyBlock, bool IsJavaRecord=false)
TokenKind
Provides a simple uniform namespace for tokens from all C languages.
Definition TokenKinds.h:33
bool isLiteral(TokenKind K)
Return true if this is a "literal" kind, like a numeric constant, string, etc.
Definition TokenKinds.h:109
Top level wrappers for InstallAPI frontend operations.
bool isLineComment(const FormatToken &FormatTok)
if(T->getSizeExpr()) TRY_TO(TraverseStmt(const_cast< Expr * >(T -> getSizeExpr())))
nullptr
This class represents a compute construct, representing a 'Kind' of ‘parallel’, 'serial',...
@ Default
Set to the current date and time.
const FunctionProtoType * T
@ Type
The name was classified as a type.
Definition Sema.h:558
bool continuesLineComment(const FormatToken &FormatTok, const FormatToken *Previous, const FormatToken *MinColumnToken)
@ Parens
New-expression has a C++98 paren-delimited initializer.
Definition ExprCXX.h:2249
#define false
Definition stdbool.h:26
@ LK_C
Should be used for C.
Definition Format.h:3826
@ LK_Proto
Should be used for Protocol Buffers
Definition Format.h:3840
@ IEBS_AfterExternBlock
Backwards compatible with AfterExternBlock's indenting.
Definition Format.h:3274
@ IEBS_Indent
Indents extern blocks.
Definition Format.h:3288
@ PPDIS_BeforeHash
Indents directives before the hash.
Definition Format.h:3381
@ PPDIS_None
Does not indent any directives.
Definition Format.h:3363
@ LS_Cpp20
Parse and format as C++20.
Definition Format.h:5863
@ BWACS_Always
Always wrap braces after a control statement.
Definition Format.h:1420
@ BWACS_Never
Never wrap braces after a control statement.
Definition Format.h:1399
@ BS_Whitesmiths
Like Allman but always indent braces and line up code with braces.
Definition Format.h:2239
@ RPS_Leave
Do not remove parentheses.
Definition Format.h:4811
@ RPS_ReturnStatement
Also remove parentheses enclosing the expression in a return/co_return statement.
Definition Format.h:4826
@ NI_All
Indent in all namespaces.
Definition Format.h:4008
@ NI_Inner
Indent only in inner namespaces (nested in other namespaces).
Definition Format.h:3998
@ IGLS_OuterIndent
Indent goto labels to the enclosing block (previous indenting level).
Definition Format.h:3320
@ IGLS_NoIndent
Do not indent goto labels.
Definition Format.h:3308
Encapsulates keywords that are context sensitive or for languages not properly supported by Clang's l...
IdentifierInfo * kw_instanceof
IdentifierInfo * kw_implements
IdentifierInfo * kw_override
IdentifierInfo * kw_await
IdentifierInfo * kw_extends
IdentifierInfo * kw_async
IdentifierInfo * kw_from
IdentifierInfo * kw_abstract
IdentifierInfo * kw_var
IdentifierInfo * kw_interface
IdentifierInfo * kw_function
IdentifierInfo * kw_yield
IdentifierInfo * kw_where
IdentifierInfo * kw_throws
IdentifierInfo * kw_let
IdentifierInfo * kw_import
IdentifierInfo * kw_finally
Represents a complete lambda introducer.
Definition DeclSpec.h:2884
The FormatStyle is used to configure the formatting to follow specific guidelines.
Definition Format.h:56
@ LK_Proto
Should be used for Protocol Buffers
Definition Format.h:3840
@ RCS_Always
Apply indentation rules and reflow long comments into new lines, trying to obey the ColumnLimit.
Definition Format.h:4718
@ SRS_Empty
Only merge empty records.
Definition Format.h:1088
@ BS_Whitesmiths
Like Allman but always indent braces and line up code with braces.
Definition Format.h:2239
A wrapper around a Token storing information about the whitespace characters preceding it.
bool Optional
Is optional and can be removed.
bool isNot(T Kind) const
StringRef TokenText
The raw text of the token.
bool isNoneOf(Ts... Ks) const
unsigned NewlinesBefore
The number of newlines immediately before the Token.
bool is(tok::TokenKind Kind) const
bool isOneOf(A K1, B K2) const
unsigned IsFirst
Indicates that this is the first token of the file.
FormatToken * MatchingParen
If this is a bracket, this points to the matching one.
FormatToken * Previous
The previous token in the unwrapped line.
An unwrapped line is a sequence of Token, that we would like to put on a single line if there was no ...