1//===--- UnwrappedLineParser.cpp - Format C++ code ------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8///
9/// \file
10/// This file contains the implementation of the UnwrappedLineParser,
11/// which turns a stream of tokens into UnwrappedLines.
12///
13//===----------------------------------------------------------------------===//
14
15#include "UnwrappedLineParser.h"
16#include "FormatToken.h"
17#include "FormatTokenSource.h"
18#include "Macros.h"
19#include "TokenAnnotator.h"
20#include "clang/Basic/TokenKinds.h"
21#include "llvm/ADT/STLExtras.h"
22#include "llvm/ADT/StringRef.h"
23#include "llvm/Support/Debug.h"
24#include "llvm/Support/raw_os_ostream.h"
25#include "llvm/Support/raw_ostream.h"
26
27#include <utility>
28
29#define DEBUG_TYPE "format-parser"
30
31namespace clang {
32namespace format {
33
34namespace {
35
36void printLine(llvm::raw_ostream &OS, const UnwrappedLine &Line,
37 StringRef Prefix = "", bool PrintText = false) {
38 OS << Prefix << "Line(" << Line.Level << ", FSC=" << Line.FirstStartColumn
39 << ")" << (Line.InPPDirective ? " MACRO" : "") << ": ";
40 bool NewLine = false;
41 for (std::list<UnwrappedLineNode>::const_iterator I = Line.Tokens.begin(),
42 E = Line.Tokens.end();
43 I != E; ++I) {
44 if (NewLine) {
45 OS << Prefix;
46 NewLine = false;
47 }
48 OS << I->Tok->Tok.getName() << "["
49 << "T=" << (unsigned)I->Tok->getType()
50 << ", OC=" << I->Tok->OriginalColumn << ", \"" << I->Tok->TokenText
51 << "\"] ";
52 for (const auto *CI = I->Children.begin(), *CE = I->Children.end();
53 CI != CE; ++CI) {
54 OS << "\n";
55 printLine(OS, Line: *CI, Prefix: (Prefix + " ").str());
56 NewLine = true;
57 }
58 }
59 if (!NewLine)
60 OS << "\n";
61}
62
63[[maybe_unused]] static void printDebugInfo(const UnwrappedLine &Line) {
64 printLine(OS&: llvm::dbgs(), Line);
65}
66
67class ScopedDeclarationState {
68public:
69 ScopedDeclarationState(UnwrappedLine &Line, llvm::BitVector &Stack,
70 bool MustBeDeclaration)
71 : Line(Line), Stack(Stack) {
72 Line.MustBeDeclaration = MustBeDeclaration;
73 Stack.push_back(Val: MustBeDeclaration);
74 }
75 ~ScopedDeclarationState() {
76 Stack.pop_back();
77 if (!Stack.empty())
78 Line.MustBeDeclaration = Stack.back();
79 else
80 Line.MustBeDeclaration = true;
81 }
82
83private:
84 UnwrappedLine &Line;
85 llvm::BitVector &Stack;
86};
87
88} // end anonymous namespace
89
90std::ostream &operator<<(std::ostream &Stream, const UnwrappedLine &Line) {
91 llvm::raw_os_ostream OS(Stream);
92 printLine(OS, Line);
93 return Stream;
94}
95
96class ScopedLineState {
97public:
98 // With \c DiscardLines, the lines added while in scope are discarded.
99 ScopedLineState(UnwrappedLineParser &Parser,
100 bool SwitchToPreprocessorLines = false,
101 bool DiscardLines = false)
102 : Parser(Parser), OriginalLines(Parser.CurrentLines),
103 DiscardLines(DiscardLines) {
104 if (SwitchToPreprocessorLines)
105 Parser.CurrentLines = &Parser.PreprocessorDirectives;
106 else if (!Parser.Line->Tokens.empty())
107 Parser.CurrentLines = &Parser.Line->Tokens.back().Children;
108 OriginalNumLines = Parser.CurrentLines->size();
109 PreBlockLine = std::move(Parser.Line);
110 Parser.Line = std::make_unique<UnwrappedLine>();
111 Parser.Line->Level = PreBlockLine->Level;
112 Parser.Line->PPLevel = PreBlockLine->PPLevel;
113 Parser.Line->InPPDirective = PreBlockLine->InPPDirective;
114 Parser.Line->InMacroBody = PreBlockLine->InMacroBody;
115 Parser.Line->UnbracedBodyLevel = PreBlockLine->UnbracedBodyLevel;
116 }
117
118 ~ScopedLineState() {
119 if (!Parser.Line->Tokens.empty())
120 Parser.addUnwrappedLine();
121 assert(Parser.Line->Tokens.empty());
122 if (DiscardLines)
123 Parser.CurrentLines->truncate(N: OriginalNumLines);
124 Parser.Line = std::move(PreBlockLine);
125 if (Parser.CurrentLines == &Parser.PreprocessorDirectives)
126 Parser.PP.AtEndOfPPLine = true;
127 Parser.CurrentLines = OriginalLines;
128 }
129
130private:
131 UnwrappedLineParser &Parser;
132
133 std::unique_ptr<UnwrappedLine> PreBlockLine;
134 SmallVectorImpl<UnwrappedLine> *OriginalLines;
135 size_t OriginalNumLines;
136 bool DiscardLines;
137};
138
139class CompoundStatementIndenter {
140public:
141 CompoundStatementIndenter(UnwrappedLineParser *Parser,
142 const FormatStyle &Style, unsigned &LineLevel)
143 : CompoundStatementIndenter(Parser, LineLevel,
144 Style.BraceWrapping.AfterControlStatement ==
145 FormatStyle::BWACS_Always,
146 Style.BraceWrapping.IndentBraces) {}
147 CompoundStatementIndenter(UnwrappedLineParser *Parser, unsigned &LineLevel,
148 bool WrapBrace, bool IndentBrace)
149 : LineLevel(LineLevel), OldLineLevel(LineLevel) {
150 if (WrapBrace)
151 Parser->addUnwrappedLine();
152 if (IndentBrace)
153 ++LineLevel;
154 }
155 ~CompoundStatementIndenter() { LineLevel = OldLineLevel; }
156
157private:
158 unsigned &LineLevel;
159 unsigned OldLineLevel;
160};
161
162UnwrappedLineParser::UnwrappedLineParser(
163 SourceManager &SourceMgr, const FormatStyle &Style,
164 const AdditionalKeywords &Keywords, unsigned FirstStartColumn,
165 ArrayRef<FormatToken *> Tokens, UnwrappedLineConsumer &Callback,
166 llvm::SpecificBumpPtrAllocator<FormatToken> &Allocator,
167 IdentifierTable &IdentTable)
168 : Line(new UnwrappedLine), CurrentLines(&Lines), Style(Style),
169 IsCpp(Style.isCpp()), LangOpts(getFormattingLangOpts(Style)),
170 Keywords(Keywords), CommentPragmasRegex(Style.CommentPragmas),
171 Tokens(nullptr), Callback(Callback), AllTokens(Tokens),
172 PP(getIncludeGuardState(Style: Style.IndentPPDirectives)),
173 FirstStartColumn(FirstStartColumn),
174 Macros(Style.Macros, SourceMgr, Style, Allocator, IdentTable) {}
175
176void UnwrappedLineParser::reset() {
177 PP.BranchLevel = -1;
178 PP.IncludeGuard = getIncludeGuardState(Style: Style.IndentPPDirectives);
179 PP.IncludeGuardToken = nullptr;
180 ParsedPPDirectives.clear();
181 Line.reset(p: new UnwrappedLine);
182 CommentsBeforeNextToken.clear();
183 FormatTok = nullptr;
184 PP.AtEndOfPPLine = false;
185 IsDecltypeAutoFunction = false;
186 PreprocessorDirectives.clear();
187 CurrentLines = &Lines;
188 DeclarationScopeStack.clear();
189 NestedTooDeep.clear();
190 NestedLambdas.clear();
191 PP.Stack.clear();
192 Line->FirstStartColumn = FirstStartColumn;
193
194 if (!Unexpanded.empty())
195 for (FormatToken *Token : AllTokens)
196 Token->MacroCtx.reset();
197 CurrentExpandedLines.clear();
198 ExpandedLines.clear();
199 Unexpanded.clear();
200 InExpansion = false;
201 Reconstruct.reset();
202}
203
204void UnwrappedLineParser::parse() {
205 IndexedTokenSource TokenSource(AllTokens);
206 Line->FirstStartColumn = FirstStartColumn;
207 do {
208 LLVM_DEBUG(llvm::dbgs() << "----\n");
209 reset();
210 Tokens = &TokenSource;
211 TokenSource.reset();
212
213 readToken();
214 parseFile();
215
216 // If we found an include guard then all preprocessor directives (other than
217 // the guard) are over-indented by one.
218 if (PP.IncludeGuard == IG_Found) {
219 for (auto &Line : Lines)
220 if (Line.InPPDirective && Line.Level > 0)
221 --Line.Level;
222 }
223
224 // Create line with eof token.
225 assert(eof());
226 pushToken(Tok: FormatTok);
227 addUnwrappedLine();
228
229 // In a first run, format everything with the lines containing macro calls
230 // replaced by the expansion.
231 if (!ExpandedLines.empty()) {
232 LLVM_DEBUG(llvm::dbgs() << "Expanded lines:\n");
233 for (const auto &Line : Lines) {
234 if (!Line.Tokens.empty()) {
235 auto it = ExpandedLines.find(Val: Line.Tokens.begin()->Tok);
236 if (it != ExpandedLines.end()) {
237 for (const auto &Expanded : it->second) {
238 LLVM_DEBUG(printDebugInfo(Expanded));
239 Callback.consumeUnwrappedLine(Line: Expanded);
240 }
241 continue;
242 }
243 }
244 LLVM_DEBUG(printDebugInfo(Line));
245 Callback.consumeUnwrappedLine(Line);
246 }
247 Callback.finishRun();
248 }
249
250 LLVM_DEBUG(llvm::dbgs() << "Unwrapped lines:\n");
251 for (const UnwrappedLine &Line : Lines) {
252 LLVM_DEBUG(printDebugInfo(Line));
253 Callback.consumeUnwrappedLine(Line);
254 }
255 Callback.finishRun();
256 Lines.clear();
257 while (!PP.LevelBranchIndex.empty() &&
258 PP.LevelBranchIndex.back() + 1 >= PP.LevelBranchCount.back()) {
259 PP.LevelBranchIndex.resize(N: PP.LevelBranchIndex.size() - 1);
260 PP.LevelBranchCount.resize(N: PP.LevelBranchCount.size() - 1);
261 }
262 if (!PP.LevelBranchIndex.empty()) {
263 ++PP.LevelBranchIndex.back();
264 assert(PP.LevelBranchIndex.size() == PP.LevelBranchCount.size());
265 assert(PP.LevelBranchIndex.back() <= PP.LevelBranchCount.back());
266 }
267 } while (!PP.LevelBranchIndex.empty());
268}
269
270void UnwrappedLineParser::parseFile() {
271 // The top-level context in a file always has declarations, except for pre-
272 // processor directives and JavaScript files.
273 bool MustBeDeclaration = !Line->InPPDirective && !Style.isJavaScript();
274 ScopedDeclarationState DeclarationState(*Line, DeclarationScopeStack,
275 MustBeDeclaration);
276 if (Style.isTextProto() || (Style.isJson() && FormatTok->IsFirst))
277 parseBracedList();
278 else
279 parseLevel();
280 // Make sure to format the remaining tokens.
281 //
282 // LK_TextProto is special since its top-level is parsed as the body of a
283 // braced list, which does not necessarily have natural line separators such
284 // as a semicolon. Comments after the last entry that have been determined to
285 // not belong to that line, as in:
286 // key: value
287 // // endfile comment
288 // do not have a chance to be put on a line of their own until this point.
289 // Here we add this newline before end-of-file comments.
290 if (Style.isTextProto() && !CommentsBeforeNextToken.empty())
291 addUnwrappedLine();
292 flushComments(NewlineBeforeNext: true);
293 addUnwrappedLine();
294}
295
296void UnwrappedLineParser::parseCSharpGenericTypeConstraint() {
297 do {
298 switch (FormatTok->Tok.getKind()) {
299 case tok::l_brace:
300 case tok::semi:
301 return;
302 default:
303 if (FormatTok->is(II: Keywords.kw_where)) {
304 addUnwrappedLine();
305 nextToken();
306 parseCSharpGenericTypeConstraint();
307 break;
308 }
309 nextToken();
310 break;
311 }
312 } while (!eof());
313}
314
315void UnwrappedLineParser::parseCSharpAttribute() {
316 int UnpairedSquareBrackets = 1;
317 do {
318 switch (FormatTok->Tok.getKind()) {
319 case tok::r_square:
320 nextToken();
321 --UnpairedSquareBrackets;
322 if (UnpairedSquareBrackets == 0) {
323 addUnwrappedLine();
324 return;
325 }
326 break;
327 case tok::l_square:
328 ++UnpairedSquareBrackets;
329 nextToken();
330 break;
331 default:
332 nextToken();
333 break;
334 }
335 } while (!eof());
336}
337
338bool UnwrappedLineParser::precededByCommentOrPPDirective() const {
339 if (!Lines.empty() && Lines.back().InPPDirective)
340 return true;
341
342 const FormatToken *Previous = Tokens->getPreviousToken();
343 return Previous && Previous->is(Kind: tok::comment) &&
344 (Previous->IsMultiline || Previous->NewlinesBefore > 0);
345}
346
347/// Parses a level, that is ???.
348/// \param OpeningBrace Opening brace (\p nullptr if absent) of that level.
349/// \param IfKind The \p if statement kind in the level.
350/// \param IfLeftBrace The left brace of the \p if block in the level.
351/// \returns true if a simple block of if/else/for/while, or false otherwise.
352/// (A simple block has a single statement.)
353bool UnwrappedLineParser::parseLevel(const FormatToken *OpeningBrace,
354 IfStmtKind *IfKind,
355 FormatToken **IfLeftBrace,
356 bool *SeenExplicitAccessModifier) {
357 const bool InRequiresExpression =
358 OpeningBrace && OpeningBrace->is(TT: TT_RequiresExpressionLBrace);
359 const bool IsPrecededByCommentOrPPDirective =
360 !Style.RemoveBracesLLVM || precededByCommentOrPPDirective();
361 FormatToken *IfLBrace = nullptr;
362 bool HasDoWhile = false;
363 bool HasLabel = false;
364 unsigned StatementCount = 0;
365 bool SwitchLabelEncountered = false;
366
367 do {
368 if (FormatTok->isAttribute()) {
369 nextToken();
370 if (FormatTok->is(Kind: tok::l_paren))
371 parseParens();
372 continue;
373 }
374 tok::TokenKind Kind = FormatTok->Tok.getKind();
375 if (FormatTok->is(TT: TT_MacroBlockBegin))
376 Kind = tok::l_brace;
377 else if (FormatTok->is(TT: TT_MacroBlockEnd))
378 Kind = tok::r_brace;
379
380 auto ParseDefault = [this, OpeningBrace, IfKind, &IfLBrace, &HasDoWhile,
381 &HasLabel, &StatementCount,
382 SeenExplicitAccessModifier] {
383 if (SeenExplicitAccessModifier && !*SeenExplicitAccessModifier) {
384 const bool IsQtAccessLabel =
385 FormatTok->isOneOf(K1: Keywords.kw_signals, K2: Keywords.kw_qsignals,
386 Ks: Keywords.kw_slots, Ks: Keywords.kw_qslots) &&
387 Tokens->peekNextToken(/*SkipComment=*/true)->is(Kind: tok::colon);
388 if (FormatTok->isAccessSpecifierKeyword() || IsQtAccessLabel) {
389 ++Line->Level;
390 *SeenExplicitAccessModifier = true;
391 }
392 }
393 parseStructuralElement(OpeningBrace, IfKind, IfLeftBrace: &IfLBrace,
394 HasDoWhile: HasDoWhile ? nullptr : &HasDoWhile,
395 HasLabel: HasLabel ? nullptr : &HasLabel);
396 ++StatementCount;
397 assert(StatementCount > 0 && "StatementCount overflow!");
398 };
399
400 switch (Kind) {
401 case tok::comment:
402 nextToken();
403 addUnwrappedLine();
404 break;
405 case tok::l_brace:
406 if (InRequiresExpression) {
407 FormatTok->setFinalizedType(TT_CompoundRequirementLBrace);
408 } else if (FormatTok->Previous &&
409 FormatTok->Previous->ClosesRequiresClause) {
410 // We need the 'default' case here to correctly parse a function
411 // l_brace.
412 ParseDefault();
413 continue;
414 }
415 if (!InRequiresExpression && FormatTok->isNot(Kind: TT_MacroBlockBegin)) {
416 if (tryToParseBracedList())
417 continue;
418 FormatTok->setFinalizedType(TT_BlockLBrace);
419 }
420 parseBlock();
421 ++StatementCount;
422 assert(StatementCount > 0 && "StatementCount overflow!");
423 addUnwrappedLine();
424 break;
425 case tok::r_brace:
426 if (OpeningBrace) {
427 if (!Style.RemoveBracesLLVM || Line->InPPDirective ||
428 OpeningBrace->isNoneOf(Ks: TT_ControlStatementLBrace, Ks: TT_ElseLBrace)) {
429 return false;
430 }
431 if (FormatTok->isNot(Kind: tok::r_brace) || StatementCount != 1 || HasLabel ||
432 HasDoWhile || IsPrecededByCommentOrPPDirective ||
433 precededByCommentOrPPDirective()) {
434 return false;
435 }
436 const FormatToken *Next = Tokens->peekNextToken();
437 if (Next->is(Kind: tok::comment) && Next->NewlinesBefore == 0)
438 return false;
439 if (IfLeftBrace)
440 *IfLeftBrace = IfLBrace;
441 return true;
442 }
443 nextToken();
444 addUnwrappedLine();
445 break;
446 case tok::kw_default: {
447 unsigned StoredPosition = Tokens->getPosition();
448 auto *Next = Tokens->getNextNonComment();
449 FormatTok = Tokens->setPosition(StoredPosition);
450 if (Next->isNoneOf(Ks: tok::colon, Ks: tok::arrow)) {
451 // default not followed by `:` or `->` is not a case label; treat it
452 // like an identifier.
453 parseStructuralElement();
454 break;
455 }
456 // Else, if it is 'default:', fall through to the case handling.
457 [[fallthrough]];
458 }
459 case tok::kw_case:
460 if (Style.Language == FormatStyle::LK_Proto || Style.isVerilog() ||
461 (Style.isJavaScript() && Line->MustBeDeclaration)) {
462 // Proto: there are no switch/case statements
463 // Verilog: Case labels don't have this word. We handle case
464 // labels including default in TokenAnnotator.
465 // JavaScript: A 'case: string' style field declaration.
466 ParseDefault();
467 break;
468 }
469 if (!SwitchLabelEncountered &&
470 (Style.IndentCaseLabels ||
471 (OpeningBrace && OpeningBrace->is(TT: TT_SwitchExpressionLBrace)) ||
472 (Line->InPPDirective && Line->Level == 1))) {
473 ++Line->Level;
474 }
475 SwitchLabelEncountered = true;
476 parseStructuralElement();
477 break;
478 case tok::l_square:
479 if (Style.isCSharp()) {
480 nextToken();
481 parseCSharpAttribute();
482 break;
483 }
484 if (handleCppAttributes())
485 break;
486 [[fallthrough]];
487 default:
488 ParseDefault();
489 break;
490 }
491 } while (!eof());
492
493 return false;
494}
495
496void UnwrappedLineParser::calculateBraceTypes(bool ExpectClassBody) {
497 // We'll parse forward through the tokens until we hit
498 // a closing brace or eof - note that getNextToken() will
499 // parse macros, so this will magically work inside macro
500 // definitions, too.
501 unsigned StoredPosition = Tokens->getPosition();
502 FormatToken *Tok = FormatTok;
503 const FormatToken *PrevTok = Tok->Previous;
504 // Keep a stack of positions of lbrace tokens. We will
505 // update information about whether an lbrace starts a
506 // braced init list or a different block during the loop.
507 struct StackEntry {
508 FormatToken *Tok;
509 const FormatToken *PrevTok;
510 };
511 SmallVector<StackEntry, 8> LBraceStack;
512 assert(Tok->is(tok::l_brace));
513
514 do {
515 auto *NextTok = Tokens->getNextNonComment();
516
517 if (!Line->InMacroBody && !Style.isTableGen()) {
518 // Skip PPDirective lines (except macro definitions) and comments.
519 while (NextTok->is(Kind: tok::hash)) {
520 NextTok = Tokens->getNextToken();
521 if (NextTok->isOneOf(K1: tok::pp_not_keyword, K2: tok::pp_define))
522 break;
523 do {
524 NextTok = Tokens->getNextToken();
525 } while (!NextTok->HasUnescapedNewline && NextTok->isNot(Kind: tok::eof));
526
527 while (NextTok->is(Kind: tok::comment))
528 NextTok = Tokens->getNextToken();
529 }
530 }
531
532 switch (Tok->Tok.getKind()) {
533 case tok::l_brace:
534 if (Style.isJavaScript() && PrevTok) {
535 if (PrevTok->isOneOf(K1: tok::colon, K2: tok::less)) {
536 // A ':' indicates this code is in a type, or a braced list
537 // following a label in an object literal ({a: {b: 1}}).
538 // A '<' could be an object used in a comparison, but that is nonsense
539 // code (can never return true), so more likely it is a generic type
540 // argument (`X<{a: string; b: number}>`).
541 // The code below could be confused by semicolons between the
542 // individual members in a type member list, which would normally
543 // trigger BK_Block. In both cases, this must be parsed as an inline
544 // braced init.
545 Tok->setBlockKind(BK_BracedInit);
546 } else if (PrevTok->is(Kind: tok::r_paren)) {
547 // `) { }` can only occur in function or method declarations in JS.
548 Tok->setBlockKind(BK_Block);
549 }
550 } else if (Style.isJava() && PrevTok && PrevTok->is(Kind: tok::arrow)) {
551 Tok->setBlockKind(BK_Block);
552 } else {
553 Tok->setBlockKind(BK_Unknown);
554 }
555 LBraceStack.push_back(Elt: {.Tok: Tok, .PrevTok: PrevTok});
556 break;
557 case tok::r_brace:
558 if (LBraceStack.empty())
559 break;
560 if (auto *LBrace = LBraceStack.back().Tok; LBrace->is(BBK: BK_Unknown)) {
561 bool ProbablyBracedList = false;
562 if (Style.Language == FormatStyle::LK_Proto) {
563 ProbablyBracedList = NextTok->isOneOf(K1: tok::comma, K2: tok::r_square);
564 } else if (LBrace->isNot(Kind: TT_EnumLBrace)) {
565 // Using OriginalColumn to distinguish between ObjC methods and
566 // binary operators is a bit hacky.
567 bool NextIsObjCMethod = NextTok->isOneOf(K1: tok::plus, K2: tok::minus) &&
568 NextTok->OriginalColumn == 0;
569
570 // Try to detect a braced list. Note that regardless how we mark inner
571 // braces here, we will overwrite the BlockKind later if we parse a
572 // braced list (where all blocks inside are by default braced lists),
573 // or when we explicitly detect blocks (for example while parsing
574 // lambdas).
575
576 // If we already marked the opening brace as braced list, the closing
577 // must also be part of it.
578 ProbablyBracedList = LBrace->is(TT: TT_BracedListLBrace);
579
580 ProbablyBracedList = ProbablyBracedList ||
581 (Style.isJavaScript() &&
582 NextTok->isOneOf(K1: Keywords.kw_of, K2: Keywords.kw_in,
583 Ks: Keywords.kw_as));
584 ProbablyBracedList =
585 ProbablyBracedList ||
586 (IsCpp && (PrevTok->Tok.isLiteral() ||
587 NextTok->isOneOf(K1: tok::l_paren, K2: tok::arrow)));
588
589 // If there is a comma, or right paren after the closing brace, we
590 // assume this is a braced initializer list.
591 // FIXME: Some of these do not apply to JS, e.g. "} {" can never be a
592 // braced list in JS.
593 ProbablyBracedList =
594 ProbablyBracedList ||
595 NextTok->isOneOf(K1: tok::comma, K2: tok::period, Ks: tok::colon,
596 Ks: tok::r_paren, Ks: tok::r_square, Ks: tok::ellipsis);
597
598 // Distinguish between braced list in a constructor initializer list
599 // followed by constructor body, or just adjacent blocks.
600 ProbablyBracedList =
601 ProbablyBracedList ||
602 (NextTok->is(Kind: tok::l_brace) && LBraceStack.back().PrevTok &&
603 LBraceStack.back().PrevTok->isOneOf(K1: tok::identifier,
604 K2: tok::greater));
605
606 ProbablyBracedList =
607 ProbablyBracedList ||
608 (NextTok->is(Kind: tok::identifier) &&
609 PrevTok->isNoneOf(Ks: tok::semi, Ks: tok::r_brace, Ks: tok::l_brace));
610
611 ProbablyBracedList = ProbablyBracedList ||
612 (NextTok->is(Kind: tok::semi) &&
613 (!ExpectClassBody || LBraceStack.size() != 1));
614
615 ProbablyBracedList =
616 ProbablyBracedList ||
617 (NextTok->isBinaryOperator() && !NextIsObjCMethod);
618
619 if (!Style.isCSharp() && NextTok->is(Kind: tok::l_square)) {
620 // We can have an array subscript after a braced init
621 // list, but C++11 attributes are expected after blocks.
622 NextTok = Tokens->getNextToken();
623 ProbablyBracedList = NextTok->isNot(Kind: tok::l_square);
624 }
625
626 // Cpp macro definition body that is a nonempty braced list or block:
627 if (IsCpp && Line->InMacroBody && PrevTok != FormatTok &&
628 !FormatTok->Previous && NextTok->is(Kind: tok::eof) &&
629 // A statement can end with only `;` (simple statement), a block
630 // closing brace (compound statement), or `:` (label statement).
631 // If PrevTok is a block opening brace, Tok ends an empty block.
632 PrevTok->isNoneOf(Ks: tok::semi, Ks: BK_Block, Ks: tok::colon)) {
633 ProbablyBracedList = true;
634 }
635 }
636 const auto BlockKind = ProbablyBracedList ? BK_BracedInit : BK_Block;
637 Tok->setBlockKind(BlockKind);
638 LBrace->setBlockKind(BlockKind);
639 }
640 LBraceStack.pop_back();
641 break;
642 case tok::identifier:
643 if (Tok->isNot(Kind: TT_StatementMacro))
644 break;
645 [[fallthrough]];
646 case tok::at:
647 case tok::semi:
648 case tok::kw_if:
649 case tok::kw_while:
650 case tok::kw_for:
651 case tok::kw_switch:
652 case tok::kw_try:
653 case tok::kw___try:
654 if (!LBraceStack.empty() && LBraceStack.back().Tok->is(BBK: BK_Unknown))
655 LBraceStack.back().Tok->setBlockKind(BK_Block);
656 break;
657 default:
658 break;
659 }
660
661 PrevTok = Tok;
662 Tok = NextTok;
663 } while (Tok->isNot(Kind: tok::eof) && !LBraceStack.empty());
664
665 // Assume other blocks for all unclosed opening braces.
666 for (const auto &Entry : LBraceStack)
667 if (Entry.Tok->is(BBK: BK_Unknown))
668 Entry.Tok->setBlockKind(BK_Block);
669
670 FormatTok = Tokens->setPosition(StoredPosition);
671}
672
673// Sets the token type of the directly previous right brace.
674void UnwrappedLineParser::setPreviousRBraceType(TokenType Type) {
675 if (auto Prev = FormatTok->getPreviousNonComment();
676 Prev && Prev->is(Kind: tok::r_brace)) {
677 Prev->setFinalizedType(Type);
678 }
679}
680
681template <class T>
682static inline void hash_combine(std::size_t &seed, const T &v) {
683 std::hash<T> hasher;
684 seed ^= hasher(v) + 0x9e3779b9 + (seed << 6) + (seed >> 2);
685}
686
687size_t UnwrappedLineParser::computePPHash() const {
688 size_t h = 0;
689 for (const auto &i : PP.Stack) {
690 hash_combine(seed&: h, v: size_t(i.Kind));
691 hash_combine(seed&: h, v: i.Line);
692 }
693 return h;
694}
695
696// Checks whether \p ParsedLine might fit on a single line. If \p OpeningBrace
697// is not null, subtracts its length (plus the preceding space) when computing
698// the length of \p ParsedLine. We must clone the tokens of \p ParsedLine before
699// running the token annotator on it so that we can restore them afterward.
700bool UnwrappedLineParser::mightFitOnOneLine(
701 UnwrappedLine &ParsedLine, const FormatToken *OpeningBrace) const {
702 const auto ColumnLimit = Style.ColumnLimit;
703 if (ColumnLimit == 0)
704 return true;
705
706 auto &Tokens = ParsedLine.Tokens;
707 assert(!Tokens.empty());
708
709 const auto *LastToken = Tokens.back().Tok;
710 assert(LastToken);
711
712 SmallVector<UnwrappedLineNode> SavedTokens(Tokens.size());
713
714 int Index = 0;
715 for (const auto &Token : Tokens) {
716 assert(Token.Tok);
717 auto &SavedToken = SavedTokens[Index++];
718 SavedToken.Tok = new FormatToken;
719 SavedToken.Tok->copyFrom(Tok: *Token.Tok);
720 SavedToken.Children = std::move(Token.Children);
721 }
722
723 AnnotatedLine Line(ParsedLine);
724 assert(Line.Last == LastToken);
725
726 TokenAnnotator Annotator(Style, Keywords);
727 Annotator.annotate(Line);
728 Annotator.calculateFormattingInformation(Line);
729
730 auto Length = LastToken->TotalLength;
731 if (OpeningBrace) {
732 assert(OpeningBrace != Tokens.front().Tok);
733 if (auto Prev = OpeningBrace->Previous;
734 Prev && Prev->TotalLength + ColumnLimit == OpeningBrace->TotalLength) {
735 Length -= ColumnLimit;
736 }
737 Length -= OpeningBrace->TokenText.size() + 1;
738 }
739
740 if (const auto *FirstToken = Line.First; FirstToken->is(Kind: tok::r_brace)) {
741 assert(!OpeningBrace || OpeningBrace->is(TT_ControlStatementLBrace));
742 Length -= FirstToken->TokenText.size() + 1;
743 }
744
745 Index = 0;
746 for (auto &Token : Tokens) {
747 const auto &SavedToken = SavedTokens[Index++];
748 Token.Tok->copyFrom(Tok: *SavedToken.Tok);
749 Token.Children = std::move(SavedToken.Children);
750 delete SavedToken.Tok;
751 }
752
753 // If these change PPLevel needs to be used for get correct indentation.
754 assert(!Line.InMacroBody);
755 assert(!Line.InPPDirective);
756 return Line.Level * Style.IndentWidth + Length <= ColumnLimit;
757}
758
759FormatToken *UnwrappedLineParser::parseBlock(
760 bool MustBeDeclaration, unsigned AddLevels, bool MunchSemi, bool KeepBraces,
761 IfStmtKind *IfKind, bool UnindentWhitesmithsBraces,
762 bool IndentAfterExplicitAccessModifier) {
763 auto HandleVerilogBlockLabel = [this]() {
764 // ":" name
765 if (Style.isVerilog() && FormatTok->is(Kind: tok::colon)) {
766 nextToken();
767 if (Keywords.isVerilogIdentifier(Tok: *FormatTok))
768 nextToken();
769 }
770 };
771
772 // Whether this is a Verilog-specific block that has a special header like a
773 // module.
774 const bool VerilogHierarchy =
775 Style.isVerilog() && Keywords.isVerilogHierarchy(Tok: *FormatTok);
776 assert((FormatTok->isOneOf(tok::l_brace, TT_MacroBlockBegin) ||
777 (Style.isVerilog() &&
778 (Keywords.isVerilogBegin(*FormatTok) || VerilogHierarchy))) &&
779 "'{' or macro block token expected");
780 FormatToken *Tok = FormatTok;
781 const bool FollowedByComment = Tokens->peekNextToken()->is(Kind: tok::comment);
782 auto Index = CurrentLines->size();
783 const bool MacroBlock = FormatTok->is(TT: TT_MacroBlockBegin);
784 FormatTok->setBlockKind(BK_Block);
785
786 const bool IsWhitesmiths =
787 Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths;
788
789 // For Whitesmiths mode, jump to the next level prior to skipping over the
790 // braces.
791 if (!VerilogHierarchy && AddLevels > 0 && IsWhitesmiths)
792 ++Line->Level;
793
794 size_t PPStartHash = computePPHash();
795
796 const unsigned InitialLevel = Line->Level;
797 if (VerilogHierarchy) {
798 AddLevels += parseVerilogHierarchyHeader();
799 } else {
800 nextToken(/*LevelDifference=*/AddLevels);
801 HandleVerilogBlockLabel();
802 }
803
804 // Bail out if there are too many levels. Otherwise, the stack might overflow.
805 if (Line->Level > 300)
806 return nullptr;
807
808 if (MacroBlock && FormatTok->is(Kind: tok::l_paren))
809 parseParens();
810
811 size_t NbPreprocessorDirectives =
812 !parsingPPDirective() ? PreprocessorDirectives.size() : 0;
813 addUnwrappedLine();
814 size_t OpeningLineIndex =
815 CurrentLines->empty()
816 ? (UnwrappedLine::kInvalidIndex)
817 : (CurrentLines->size() - 1 - NbPreprocessorDirectives);
818
819 // Whitesmiths is weird here. The brace needs to be indented for the namespace
820 // block, but the block itself may not be indented depending on the style
821 // settings. This allows the format to back up one level in those cases.
822 if (UnindentWhitesmithsBraces)
823 --Line->Level;
824
825 ScopedDeclarationState DeclarationState(*Line, DeclarationScopeStack,
826 MustBeDeclaration);
827
828 // Whitesmiths logic has already added a level by this point, so avoid
829 // adding it twice.
830 if (AddLevels > 0u)
831 Line->Level += AddLevels - (IsWhitesmiths ? 1 : 0);
832
833 FormatToken *IfLBrace = nullptr;
834 bool SeenExplicitAccessModifier = false;
835 const bool SimpleBlock =
836 parseLevel(OpeningBrace: Tok, IfKind, IfLeftBrace: &IfLBrace,
837 SeenExplicitAccessModifier: IndentAfterExplicitAccessModifier ? &SeenExplicitAccessModifier
838 : nullptr);
839
840 if (eof())
841 return IfLBrace;
842
843 if (MacroBlock ? FormatTok->isNot(Kind: TT_MacroBlockEnd)
844 : FormatTok->isNot(Kind: tok::r_brace)) {
845 Line->Level = InitialLevel;
846 FormatTok->setBlockKind(BK_Block);
847 return IfLBrace;
848 }
849
850 if (FormatTok->is(Kind: tok::r_brace)) {
851 FormatTok->setBlockKind(BK_Block);
852 if (Tok->is(TT: TT_NamespaceLBrace))
853 FormatTok->setFinalizedType(TT_NamespaceRBrace);
854 }
855
856 const bool IsFunctionRBrace =
857 FormatTok->is(Kind: tok::r_brace) && Tok->is(TT: TT_FunctionLBrace);
858
859 auto RemoveBraces = [=]() mutable {
860 if (!SimpleBlock)
861 return false;
862 assert(Tok->isOneOf(TT_ControlStatementLBrace, TT_ElseLBrace));
863 assert(FormatTok->is(tok::r_brace));
864 const bool WrappedOpeningBrace = !Tok->Previous;
865 if (WrappedOpeningBrace && FollowedByComment)
866 return false;
867 const bool HasRequiredIfBraces = IfLBrace && !IfLBrace->Optional;
868 if (KeepBraces && !HasRequiredIfBraces)
869 return false;
870 if (Tok->isNot(Kind: TT_ElseLBrace) || !HasRequiredIfBraces) {
871 const FormatToken *Previous = Tokens->getPreviousToken();
872 assert(Previous);
873 if (Previous->is(Kind: tok::r_brace) && !Previous->Optional)
874 return false;
875 }
876 assert(!CurrentLines->empty());
877 auto &LastLine = CurrentLines->back();
878 if (LastLine.Level == InitialLevel + 1 && !mightFitOnOneLine(ParsedLine&: LastLine))
879 return false;
880 if (Tok->is(TT: TT_ElseLBrace))
881 return true;
882 if (WrappedOpeningBrace) {
883 assert(Index > 0);
884 --Index; // The line above the wrapped l_brace.
885 Tok = nullptr;
886 }
887 return mightFitOnOneLine(ParsedLine&: (*CurrentLines)[Index], OpeningBrace: Tok);
888 };
889 if (RemoveBraces()) {
890 Tok->MatchingParen = FormatTok;
891 FormatTok->MatchingParen = Tok;
892 }
893
894 size_t PPEndHash = computePPHash();
895
896 if (SeenExplicitAccessModifier)
897 ++AddLevels;
898 // Munch the closing brace.
899 nextToken(/*LevelDifference=*/-AddLevels);
900
901 // When this is a function block and there is an unnecessary semicolon
902 // afterwards then mark it as optional (so the RemoveSemi pass can get rid of
903 // it later).
904 if (Style.RemoveSemicolon && IsFunctionRBrace) {
905 while (FormatTok->is(Kind: tok::semi)) {
906 FormatTok->Optional = true;
907 nextToken();
908 }
909 }
910
911 HandleVerilogBlockLabel();
912
913 if (MacroBlock && FormatTok->is(Kind: tok::l_paren))
914 parseParens();
915
916 Line->Level = InitialLevel;
917
918 if (FormatTok->is(Kind: tok::kw_noexcept)) {
919 // A noexcept in a requires expression.
920 nextToken();
921 }
922
923 if (FormatTok->is(Kind: tok::arrow)) {
924 // Following the } or noexcept we can find a trailing return type arrow
925 // as part of an implicit conversion constraint.
926 nextToken();
927 parseStructuralElement();
928 }
929
930 if (MunchSemi && FormatTok->is(Kind: tok::semi))
931 nextToken();
932
933 if (PPStartHash == PPEndHash) {
934 Line->MatchingOpeningBlockLineIndex = OpeningLineIndex;
935 if (OpeningLineIndex != UnwrappedLine::kInvalidIndex) {
936 // Update the opening line to add the forward reference as well
937 (*CurrentLines)[OpeningLineIndex].MatchingClosingBlockLineIndex =
938 CurrentLines->size() - 1;
939 }
940 }
941
942 return IfLBrace;
943}
944
945static bool isGoogScope(const UnwrappedLine &Line) {
946 // FIXME: Closure-library specific stuff should not be hard-coded but be
947 // configurable.
948 if (Line.Tokens.size() < 4)
949 return false;
950 auto I = Line.Tokens.begin();
951 if (I->Tok->TokenText != "goog")
952 return false;
953 ++I;
954 if (I->Tok->isNot(Kind: tok::period))
955 return false;
956 ++I;
957 if (I->Tok->TokenText != "scope")
958 return false;
959 ++I;
960 return I->Tok->is(Kind: tok::l_paren);
961}
962
963static bool isIIFE(const UnwrappedLine &Line,
964 const AdditionalKeywords &Keywords) {
965 // Look for the start of an immediately invoked anonymous function.
966 // https://en.wikipedia.org/wiki/Immediately-invoked_function_expression
967 // This is commonly done in JavaScript to create a new, anonymous scope.
968 // Example: (function() { ... })()
969 if (Line.Tokens.size() < 3)
970 return false;
971 auto I = Line.Tokens.begin();
972 if (I->Tok->isNot(Kind: tok::l_paren))
973 return false;
974 ++I;
975 if (I->Tok->isNot(Kind: Keywords.kw_function))
976 return false;
977 ++I;
978 return I->Tok->is(Kind: tok::l_paren);
979}
980
981static bool ShouldBreakBeforeBrace(const FormatStyle &Style,
982 const FormatToken &InitialToken,
983 bool IsEmptyBlock,
984 bool IsJavaRecord = false) {
985 if (IsJavaRecord)
986 return Style.BraceWrapping.AfterClass;
987
988 tok::TokenKind Kind = InitialToken.Tok.getKind();
989 if (InitialToken.is(TT: TT_NamespaceMacro))
990 Kind = tok::kw_namespace;
991
992 const bool WrapRecordAllowed =
993 !IsEmptyBlock ||
994 Style.AllowShortRecordOnASingleLine < FormatStyle::SRS_Empty ||
995 Style.BraceWrapping.SplitEmptyRecord;
996
997 switch (Kind) {
998 case tok::kw_namespace:
999 return Style.BraceWrapping.AfterNamespace;
1000 case tok::kw_class:
1001 return Style.BraceWrapping.AfterClass && WrapRecordAllowed;
1002 case tok::kw_union:
1003 return Style.BraceWrapping.AfterUnion && WrapRecordAllowed;
1004 case tok::kw_struct:
1005 return Style.BraceWrapping.AfterStruct && WrapRecordAllowed;
1006 case tok::kw_enum:
1007 return Style.BraceWrapping.AfterEnum;
1008 default:
1009 return false;
1010 }
1011}
1012
1013void UnwrappedLineParser::parseChildBlock() {
1014 assert(FormatTok->is(tok::l_brace));
1015 FormatTok->setBlockKind(BK_Block);
1016 const FormatToken *OpeningBrace = FormatTok;
1017 nextToken();
1018 {
1019 bool SkipIndent = (Style.isJavaScript() &&
1020 (isGoogScope(Line: *Line) || isIIFE(Line: *Line, Keywords)));
1021 ScopedLineState LineState(*this);
1022 ScopedDeclarationState DeclarationState(*Line, DeclarationScopeStack,
1023 /*MustBeDeclaration=*/false);
1024 Line->Level += SkipIndent ? 0 : 1;
1025 parseLevel(OpeningBrace);
1026 flushComments(NewlineBeforeNext: isOnNewLine(FormatTok: *FormatTok));
1027 Line->Level -= SkipIndent ? 0 : 1;
1028 }
1029 nextToken();
1030}
1031
1032void UnwrappedLineParser::parsePPDirective() {
1033 assert(FormatTok->is(tok::hash) && "'#' expected");
1034 ScopedMacroState MacroState(*Line, Tokens, FormatTok);
1035
1036 nextToken();
1037
1038 if (!FormatTok->Tok.getIdentifierInfo()) {
1039 parsePPUnknown();
1040 return;
1041 }
1042
1043 switch (FormatTok->Tok.getIdentifierInfo()->getPPKeywordID()) {
1044 case tok::pp_define:
1045 parsePPDefine();
1046 return;
1047 case tok::pp_if:
1048 parsePPIf(/*IfDef=*/false);
1049 break;
1050 case tok::pp_ifdef:
1051 case tok::pp_ifndef:
1052 parsePPIf(/*IfDef=*/true);
1053 break;
1054 case tok::pp_else:
1055 case tok::pp_elifdef:
1056 case tok::pp_elifndef:
1057 case tok::pp_elif:
1058 parsePPElse();
1059 break;
1060 case tok::pp_endif:
1061 parsePPEndIf();
1062 break;
1063 case tok::pp_pragma:
1064 parsePPPragma();
1065 break;
1066 case tok::pp_error:
1067 case tok::pp_warning:
1068 nextToken();
1069 if (!eof() && Style.isCpp())
1070 FormatTok->setFinalizedType(TT_AfterPPDirective);
1071 [[fallthrough]];
1072 default:
1073 parsePPUnknown();
1074 break;
1075 }
1076}
1077
1078void UnwrappedLineParser::conditionalCompilationCondition(bool Unreachable) {
1079 size_t Line = CurrentLines->size();
1080 if (CurrentLines == &PreprocessorDirectives)
1081 Line += Lines.size();
1082
1083 if (Unreachable ||
1084 (!PP.Stack.empty() && PP.Stack.back().Kind == PP_Unreachable)) {
1085 PP.Stack.push_back(Elt: {PP_Unreachable, Line});
1086 } else {
1087 PP.Stack.push_back(Elt: {PP_Conditional, Line});
1088 }
1089}
1090
1091void UnwrappedLineParser::conditionalCompilationStart(bool Unreachable) {
1092 ++PP.BranchLevel;
1093 assert(PP.BranchLevel >= 0 &&
1094 PP.BranchLevel <= (int)PP.LevelBranchIndex.size());
1095 if (PP.BranchLevel == (int)PP.LevelBranchIndex.size()) {
1096 PP.LevelBranchIndex.push_back(Elt: 0);
1097 PP.LevelBranchCount.push_back(Elt: 0);
1098 }
1099 PP.ChainBranchIndex.push(x: Unreachable ? -1 : 0);
1100 bool Skip = PP.LevelBranchIndex[PP.BranchLevel] > 0;
1101 conditionalCompilationCondition(Unreachable: Unreachable || Skip);
1102}
1103
1104void UnwrappedLineParser::conditionalCompilationAlternative() {
1105 if (!PP.Stack.empty())
1106 PP.Stack.pop_back();
1107 assert(PP.BranchLevel < (int)PP.LevelBranchIndex.size());
1108 if (!PP.ChainBranchIndex.empty())
1109 ++PP.ChainBranchIndex.top();
1110 conditionalCompilationCondition(
1111 Unreachable: PP.BranchLevel >= 0 && !PP.ChainBranchIndex.empty() &&
1112 PP.LevelBranchIndex[PP.BranchLevel] != PP.ChainBranchIndex.top());
1113}
1114
1115void UnwrappedLineParser::conditionalCompilationEnd() {
1116 assert(PP.BranchLevel < (int)PP.LevelBranchIndex.size());
1117 if (PP.BranchLevel >= 0 && !PP.ChainBranchIndex.empty()) {
1118 if (PP.ChainBranchIndex.top() + 1 > PP.LevelBranchCount[PP.BranchLevel])
1119 PP.LevelBranchCount[PP.BranchLevel] = PP.ChainBranchIndex.top() + 1;
1120 }
1121 // Guard against #endif's without #if.
1122 if (PP.BranchLevel > -1)
1123 --PP.BranchLevel;
1124 if (!PP.ChainBranchIndex.empty())
1125 PP.ChainBranchIndex.pop();
1126 if (!PP.Stack.empty())
1127 PP.Stack.pop_back();
1128}
1129
1130void UnwrappedLineParser::parsePPIf(bool IfDef) {
1131 bool IfNDef = FormatTok->is(Kind: tok::pp_ifndef);
1132 nextToken();
1133 bool Unreachable = false;
1134 if (!IfDef && (FormatTok->is(Kind: tok::kw_false) || FormatTok->TokenText == "0"))
1135 Unreachable = true;
1136 if (IfDef && !IfNDef && FormatTok->TokenText == "SWIG")
1137 Unreachable = true;
1138 conditionalCompilationStart(Unreachable);
1139 FormatToken *IfCondition = FormatTok;
1140 // If there's a #ifndef on the first line, and the only lines before it are
1141 // comments, it could be an include guard.
1142 bool MaybeIncludeGuard = IfNDef;
1143 if (PP.IncludeGuard == IG_Inited && MaybeIncludeGuard) {
1144 for (auto &Line : Lines) {
1145 if (Line.Tokens.front().Tok->isNot(Kind: tok::comment)) {
1146 MaybeIncludeGuard = false;
1147 PP.IncludeGuard = IG_Rejected;
1148 break;
1149 }
1150 }
1151 }
1152 --PP.BranchLevel;
1153 parsePPUnknown();
1154 ++PP.BranchLevel;
1155 if (PP.IncludeGuard == IG_Inited && MaybeIncludeGuard) {
1156 PP.IncludeGuard = IG_IfNdefed;
1157 PP.IncludeGuardToken = IfCondition;
1158 }
1159}
1160
1161void UnwrappedLineParser::parsePPElse() {
1162 // If a potential include guard has an #else, it's not an include guard.
1163 if (PP.IncludeGuard == IG_Defined && PP.BranchLevel == 0)
1164 PP.IncludeGuard = IG_Rejected;
1165 // Don't crash when there is an #else without an #if.
1166 assert(PP.BranchLevel >= -1);
1167 if (PP.BranchLevel == -1)
1168 conditionalCompilationStart(/*Unreachable=*/true);
1169 conditionalCompilationAlternative();
1170 --PP.BranchLevel;
1171 parsePPUnknown();
1172 ++PP.BranchLevel;
1173}
1174
1175void UnwrappedLineParser::parsePPEndIf() {
1176 conditionalCompilationEnd();
1177 parsePPUnknown();
1178}
1179
1180void UnwrappedLineParser::parsePPDefine() {
1181 nextToken();
1182
1183 if (!FormatTok->Tok.getIdentifierInfo()) {
1184 PP.IncludeGuard = IG_Rejected;
1185 PP.IncludeGuardToken = nullptr;
1186 parsePPUnknown();
1187 return;
1188 }
1189
1190 bool MaybeIncludeGuard = false;
1191 if (PP.IncludeGuard == IG_IfNdefed &&
1192 PP.IncludeGuardToken->TokenText == FormatTok->TokenText) {
1193 PP.IncludeGuard = IG_Defined;
1194 PP.IncludeGuardToken = nullptr;
1195 for (auto &Line : Lines) {
1196 if (Line.Tokens.front().Tok->isNoneOf(Ks: tok::comment, Ks: tok::hash)) {
1197 PP.IncludeGuard = IG_Rejected;
1198 break;
1199 }
1200 }
1201 MaybeIncludeGuard = PP.IncludeGuard == IG_Defined;
1202 }
1203
1204 // In the context of a define, even keywords should be treated as normal
1205 // identifiers. Setting the kind to identifier is not enough, because we need
1206 // to treat additional keywords like __except as well, which are already
1207 // identifiers. Setting the identifier info to null interferes with include
1208 // guard processing above, and changes preprocessing nesting.
1209 FormatTok->Tok.setKind(tok::identifier);
1210 FormatTok->Tok.setIdentifierInfo(Keywords.kw_internal_ident_after_define);
1211 nextToken();
1212
1213 // IncludeGuard can't have a non-empty macro definition.
1214 if (MaybeIncludeGuard && !eof())
1215 PP.IncludeGuard = IG_Rejected;
1216
1217 if (FormatTok->is(Kind: tok::l_paren) && !FormatTok->hasWhitespaceBefore())
1218 parseParens();
1219 if (Style.IndentPPDirectives != FormatStyle::PPDIS_None)
1220 Line->Level += PP.BranchLevel + 1;
1221 addUnwrappedLine();
1222 ++Line->Level;
1223
1224 Line->PPLevel = PP.BranchLevel + (PP.IncludeGuard == IG_Defined ? 0 : 1);
1225 assert((int)Line->PPLevel >= 0);
1226
1227 if (eof())
1228 return;
1229
1230 Line->InMacroBody = true;
1231
1232 if (!Style.SkipMacroDefinitionBody) {
1233 // Errors during a preprocessor directive can only affect the layout of the
1234 // preprocessor directive, and thus we ignore them. An alternative approach
1235 // would be to use the same approach we use on the file level (no
1236 // re-indentation if there was a structural error) within the macro
1237 // definition.
1238 parseFile();
1239 return;
1240 }
1241
1242 for (auto *Comment : CommentsBeforeNextToken)
1243 Comment->Finalized = true;
1244
1245 do {
1246 FormatTok->Finalized = true;
1247 FormatTok = Tokens->getNextToken();
1248 } while (!eof());
1249
1250 addUnwrappedLine();
1251}
1252
1253void UnwrappedLineParser::parsePPPragma() {
1254 Line->InPragmaDirective = true;
1255 parsePPUnknown();
1256}
1257
1258void UnwrappedLineParser::parsePPUnknown() {
1259 while (!eof())
1260 nextToken();
1261 if (Style.IndentPPDirectives != FormatStyle::PPDIS_None)
1262 Line->Level += PP.BranchLevel + 1;
1263 addUnwrappedLine();
1264}
1265
1266// Here we exclude certain tokens that are not usually the first token in an
1267// unwrapped line. This is used in attempt to distinguish macro calls without
1268// trailing semicolons from other constructs split to several lines.
1269static bool tokenCanStartNewLine(const FormatToken &Tok) {
1270 // Semicolon can be a null-statement, l_square can be a start of a macro or
1271 // a C++11 attribute, but this doesn't seem to be common.
1272 return Tok.isNoneOf(Ks: tok::semi, Ks: tok::l_brace,
1273 // Tokens that can only be used as binary operators and a
1274 // part of overloaded operator names.
1275 Ks: tok::period, Ks: tok::periodstar, Ks: tok::arrow, Ks: tok::arrowstar,
1276 Ks: tok::less, Ks: tok::greater, Ks: tok::slash, Ks: tok::percent,
1277 Ks: tok::lessless, Ks: tok::greatergreater, Ks: tok::equal,
1278 Ks: tok::plusequal, Ks: tok::minusequal, Ks: tok::starequal,
1279 Ks: tok::slashequal, Ks: tok::percentequal, Ks: tok::ampequal,
1280 Ks: tok::pipeequal, Ks: tok::caretequal, Ks: tok::greatergreaterequal,
1281 Ks: tok::lesslessequal,
1282 // Colon is used in labels, base class lists, initializer
1283 // lists, range-based for loops, ternary operator, but
1284 // should never be the first token in an unwrapped line.
1285 Ks: tok::colon,
1286 // 'noexcept' is a trailing annotation.
1287 Ks: tok::kw_noexcept);
1288}
1289
1290static bool mustBeJSIdent(const AdditionalKeywords &Keywords,
1291 const FormatToken *FormatTok) {
1292 // FIXME: This returns true for C/C++ keywords like 'struct'.
1293 return FormatTok->is(Kind: tok::identifier) &&
1294 (!FormatTok->Tok.getIdentifierInfo() ||
1295 FormatTok->isNoneOf(
1296 Ks: Keywords.kw_in, Ks: Keywords.kw_of, Ks: Keywords.kw_as, Ks: Keywords.kw_async,
1297 Ks: Keywords.kw_await, Ks: Keywords.kw_yield, Ks: Keywords.kw_finally,
1298 Ks: Keywords.kw_function, Ks: Keywords.kw_import, Ks: Keywords.kw_is,
1299 Ks: Keywords.kw_let, Ks: Keywords.kw_var, Ks: tok::kw_const,
1300 Ks: Keywords.kw_abstract, Ks: Keywords.kw_extends, Ks: Keywords.kw_implements,
1301 Ks: Keywords.kw_instanceof, Ks: Keywords.kw_interface,
1302 Ks: Keywords.kw_override, Ks: Keywords.kw_throws, Ks: Keywords.kw_from));
1303}
1304
1305static bool mustBeJSIdentOrValue(const AdditionalKeywords &Keywords,
1306 const FormatToken *FormatTok) {
1307 return FormatTok->Tok.isLiteral() ||
1308 FormatTok->isOneOf(K1: tok::kw_true, K2: tok::kw_false) ||
1309 mustBeJSIdent(Keywords, FormatTok);
1310}
1311
1312// isJSDeclOrStmt returns true if |FormatTok| starts a declaration or statement
1313// when encountered after a value (see mustBeJSIdentOrValue).
1314static bool isJSDeclOrStmt(const AdditionalKeywords &Keywords,
1315 const FormatToken *FormatTok) {
1316 return FormatTok->isOneOf(
1317 K1: tok::kw_return, K2: Keywords.kw_yield,
1318 // conditionals
1319 Ks: tok::kw_if, Ks: tok::kw_else,
1320 // loops
1321 Ks: tok::kw_for, Ks: tok::kw_while, Ks: tok::kw_do, Ks: tok::kw_continue, Ks: tok::kw_break,
1322 // switch/case
1323 Ks: tok::kw_switch, Ks: tok::kw_case,
1324 // exceptions
1325 Ks: tok::kw_throw, Ks: tok::kw_try, Ks: tok::kw_catch, Ks: Keywords.kw_finally,
1326 // declaration
1327 Ks: tok::kw_const, Ks: tok::kw_class, Ks: Keywords.kw_var, Ks: Keywords.kw_let,
1328 Ks: Keywords.kw_async, Ks: Keywords.kw_function,
1329 // import/export
1330 Ks: Keywords.kw_import, Ks: tok::kw_export);
1331}
1332
1333// Checks whether a token is a type in K&R C (aka C78).
1334static bool isC78Type(const FormatToken &Tok) {
1335 return Tok.isOneOf(K1: tok::kw_char, K2: tok::kw_short, Ks: tok::kw_int, Ks: tok::kw_long,
1336 Ks: tok::kw_unsigned, Ks: tok::kw_float, Ks: tok::kw_double,
1337 Ks: tok::identifier);
1338}
1339
1340// This function checks whether a token starts the first parameter declaration
1341// in a K&R C (aka C78) function definition, e.g.:
1342// int f(a, b)
1343// short a, b;
1344// {
1345// return a + b;
1346// }
1347static bool isC78ParameterDecl(const FormatToken *Tok, const FormatToken *Next,
1348 const FormatToken *FuncName) {
1349 assert(Tok);
1350 assert(Next);
1351 assert(FuncName);
1352
1353 if (FuncName->isNot(Kind: tok::identifier))
1354 return false;
1355
1356 const FormatToken *Prev = FuncName->Previous;
1357 if (!Prev || (Prev->isNot(Kind: tok::star) && !isC78Type(Tok: *Prev)))
1358 return false;
1359
1360 if (!isC78Type(Tok: *Tok) &&
1361 Tok->isNoneOf(Ks: tok::kw_register, Ks: tok::kw_struct, Ks: tok::kw_union)) {
1362 return false;
1363 }
1364
1365 if (Next->isNot(Kind: tok::star) && !Next->Tok.getIdentifierInfo())
1366 return false;
1367
1368 Tok = Tok->Previous;
1369 if (!Tok || Tok->isNot(Kind: tok::r_paren))
1370 return false;
1371
1372 Tok = Tok->Previous;
1373 if (!Tok || Tok->isNot(Kind: tok::identifier))
1374 return false;
1375
1376 return Tok->Previous && Tok->Previous->isOneOf(K1: tok::l_paren, K2: tok::comma);
1377}
1378
1379bool UnwrappedLineParser::parseModuleDecl() {
1380 assert(IsCpp);
1381 assert(FormatTok->is(Keywords.kw_module));
1382
1383 if (Style.Language == FormatStyle::LK_C ||
1384 Style.Standard < FormatStyle::LS_Cpp20) {
1385 return false;
1386 }
1387
1388 nextToken();
1389 if (FormatTok->isNot(Kind: tok::identifier))
1390 return false;
1391
1392 for (nextToken(); FormatTok->isNoneOf(Ks: tok::semi, Ks: tok::eof); nextToken())
1393 if (FormatTok->is(Kind: tok::colon))
1394 FormatTok->setFinalizedType(TT_ModulePartitionColon);
1395
1396 nextToken();
1397 Line->IsModuleOrImportDecl = true;
1398 addUnwrappedLine();
1399 return true;
1400}
1401
1402bool UnwrappedLineParser::parseImportDecl() {
1403 assert(IsCpp);
1404 assert(FormatTok->is(Keywords.kw_import) && "'import' expected");
1405
1406 if (Style.Language == FormatStyle::LK_C ||
1407 Style.Standard < FormatStyle::LS_Cpp20) {
1408 return false;
1409 }
1410
1411 nextToken();
1412 if (FormatTok->is(Kind: tok::colon)) {
1413 FormatTok->setFinalizedType(TT_ModulePartitionColon);
1414 nextToken();
1415 }
1416 if (FormatTok->isNoneOf(Ks: tok::identifier, Ks: tok::less, Ks: tok::string_literal))
1417 return false;
1418
1419 for (; FormatTok->isNoneOf(Ks: tok::semi, Ks: tok::eof); nextToken()) {
1420 // Handle import <foo/bar.h> as we would an include statement.
1421 if (FormatTok->is(Kind: tok::less)) {
1422 for (nextToken(); FormatTok->isNoneOf(Ks: tok::greater, Ks: tok::semi, Ks: tok::eof);
1423 nextToken()) {
1424 // Mark tokens as implicit string literals, so that import <A/Foo> will
1425 // neither be broken nor have a space added.
1426 FormatTok->setFinalizedType(TT_ImplicitStringLiteral);
1427 }
1428 }
1429 }
1430
1431 nextToken();
1432 Line->IsModuleOrImportDecl = true;
1433 addUnwrappedLine();
1434 return true;
1435}
1436
1437// readTokenWithJavaScriptASI reads the next token and terminates the current
1438// line if JavaScript Automatic Semicolon Insertion must
1439// happen between the current token and the next token.
1440//
1441// This method is conservative - it cannot cover all edge cases of JavaScript,
1442// but only aims to correctly handle certain well known cases. It *must not*
1443// return true in speculative cases.
1444void UnwrappedLineParser::readTokenWithJavaScriptASI() {
1445 FormatToken *Previous = FormatTok;
1446 readToken();
1447 FormatToken *Next = FormatTok;
1448
1449 bool IsOnSameLine =
1450 CommentsBeforeNextToken.empty()
1451 ? Next->NewlinesBefore == 0
1452 : CommentsBeforeNextToken.front()->NewlinesBefore == 0;
1453 if (IsOnSameLine)
1454 return;
1455
1456 bool PreviousMustBeValue = mustBeJSIdentOrValue(Keywords, FormatTok: Previous);
1457 bool PreviousStartsTemplateExpr =
1458 Previous->is(TT: TT_TemplateString) && Previous->TokenText.ends_with(Suffix: "${");
1459 if (PreviousMustBeValue || Previous->is(Kind: tok::r_paren)) {
1460 // If the line contains an '@' sign, the previous token might be an
1461 // annotation, which can precede another identifier/value.
1462 bool HasAt = llvm::any_of(Range&: Line->Tokens, P: [](UnwrappedLineNode &LineNode) {
1463 return LineNode.Tok->is(Kind: tok::at);
1464 });
1465 if (HasAt)
1466 return;
1467 }
1468 if (Next->is(Kind: tok::exclaim) && PreviousMustBeValue)
1469 return addUnwrappedLine();
1470 bool NextMustBeValue = mustBeJSIdentOrValue(Keywords, FormatTok: Next);
1471 bool NextEndsTemplateExpr =
1472 Next->is(TT: TT_TemplateString) && Next->TokenText.starts_with(Prefix: "}");
1473 if (NextMustBeValue && !NextEndsTemplateExpr && !PreviousStartsTemplateExpr &&
1474 (PreviousMustBeValue ||
1475 Previous->isOneOf(K1: tok::r_square, K2: tok::r_paren, Ks: tok::plusplus,
1476 Ks: tok::minusminus))) {
1477 return addUnwrappedLine();
1478 }
1479 if ((PreviousMustBeValue || Previous->is(Kind: tok::r_paren)) &&
1480 isJSDeclOrStmt(Keywords, FormatTok: Next)) {
1481 return addUnwrappedLine();
1482 }
1483}
1484
1485void UnwrappedLineParser::parseStructuralElement(
1486 const FormatToken *OpeningBrace, IfStmtKind *IfKind,
1487 FormatToken **IfLeftBrace, bool *HasDoWhile, bool *HasLabel) {
1488 if (Style.isTableGen() && FormatTok->is(Kind: tok::pp_include)) {
1489 nextToken();
1490 if (FormatTok->is(Kind: tok::string_literal))
1491 nextToken();
1492 addUnwrappedLine();
1493 return;
1494 }
1495
1496 if (IsCpp) {
1497 while (FormatTok->is(Kind: tok::l_square) && handleCppAttributes()) {
1498 }
1499 } else if (Style.isVerilog()) {
1500 // Skip attributes.
1501 while (FormatTok->is(Kind: tok::l_paren) &&
1502 Tokens->peekNextToken()->is(Kind: tok::star)) {
1503 parseParens();
1504 }
1505 skipVerilogQualifiers();
1506 // Skip things that can exist before keywords like 'if' and 'case'.
1507 if (FormatTok->isOneOf(K1: Keywords.kw_priority, K2: Keywords.kw_unique,
1508 Ks: Keywords.kw_unique0)) {
1509 nextToken();
1510 }
1511
1512 if (Keywords.isVerilogStructuredProcedure(Tok: *FormatTok)) {
1513 parseForOrWhileLoop(/*HasParens=*/false);
1514 return;
1515 }
1516 if (FormatTok->isOneOf(K1: Keywords.kw_foreach, K2: Keywords.kw_repeat)) {
1517 parseForOrWhileLoop();
1518 return;
1519 }
1520 if (FormatTok->isOneOf(K1: tok::kw_restrict, K2: Keywords.kw_assert,
1521 Ks: Keywords.kw_assume, Ks: Keywords.kw_cover)) {
1522 parseIfThenElse(IfKind, /*KeepBraces=*/false, /*IsVerilogAssert=*/true);
1523 return;
1524 }
1525 }
1526
1527 // Tokens that only make sense at the beginning of a line.
1528 if (FormatTok->isAccessSpecifierKeyword()) {
1529 if (Style.isJava() || Style.isJavaScript() || Style.isCSharp())
1530 nextToken();
1531 else
1532 parseAccessSpecifier();
1533 return;
1534 }
1535 switch (FormatTok->Tok.getKind()) {
1536 case tok::kw_asm: {
1537 // Track whether to skip formatting inline asm by finalizing the tokens
1538 // in the block. Formatting is skipped inside of braces by default.
1539 // A style option could be added to also skip formatting inside parens.
1540 bool DoNotFormat = false;
1541 tok::TokenKind OpenType;
1542 tok::TokenKind CloseType;
1543 nextToken();
1544 while (FormatTok &&
1545 FormatTok->isOneOf(K1: tok::kw_volatile, K2: tok::kw_inline, Ks: tok::kw_goto)) {
1546 nextToken();
1547 }
1548 if (!FormatTok)
1549 break;
1550 if (FormatTok->is(Kind: tok::l_brace)) {
1551 FormatTok->setFinalizedType(TT_InlineASMBrace);
1552 OpenType = tok::l_brace;
1553 CloseType = tok::r_brace;
1554 DoNotFormat = true;
1555 } else if (FormatTok->is(Kind: tok::l_paren)) {
1556 OpenType = tok::l_paren;
1557 CloseType = tok::r_paren;
1558 FormatTok->setFinalizedType(TT_InlineASMParen);
1559 } else {
1560 break;
1561 }
1562 if (DoNotFormat) {
1563 FormatToken *OpenTok = FormatTok;
1564 int NestLevel = 0;
1565 nextToken();
1566 while (FormatTok && !eof()) {
1567 if (FormatTok->is(Kind: OpenType)) {
1568 ++NestLevel;
1569 } else if (FormatTok->is(Kind: CloseType)) {
1570 --NestLevel;
1571 if (NestLevel < 1) {
1572 FormatTok->setFinalizedType(OpenTok->getType());
1573 nextToken();
1574 addUnwrappedLine();
1575 break;
1576 }
1577 }
1578 FormatTok->Finalized = true;
1579 nextToken();
1580 }
1581 }
1582 break;
1583 }
1584 case tok::kw_namespace:
1585 parseNamespace();
1586 return;
1587 case tok::kw_if: {
1588 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1589 // field/method declaration.
1590 break;
1591 }
1592 FormatToken *Tok = parseIfThenElse(IfKind);
1593 if (IfLeftBrace)
1594 *IfLeftBrace = Tok;
1595 return;
1596 }
1597 case tok::kw_for:
1598 case tok::kw_while:
1599 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1600 // field/method declaration.
1601 break;
1602 }
1603 parseForOrWhileLoop();
1604 return;
1605 case tok::kw_do:
1606 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1607 // field/method declaration.
1608 break;
1609 }
1610 parseDoWhile();
1611 if (HasDoWhile)
1612 *HasDoWhile = true;
1613 return;
1614 case tok::kw_switch:
1615 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1616 // 'switch: string' field declaration.
1617 break;
1618 }
1619 parseSwitch(/*IsExpr=*/false);
1620 return;
1621 case tok::kw_default: {
1622 // In Verilog default along with other labels are handled in the next loop.
1623 if (Style.isVerilog())
1624 break;
1625 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1626 // 'default: string' field declaration.
1627 break;
1628 }
1629 auto *Default = FormatTok;
1630 nextToken();
1631 if (FormatTok->is(Kind: tok::colon)) {
1632 FormatTok->setFinalizedType(TT_CaseLabelColon);
1633 parseLabel();
1634 return;
1635 }
1636 if (FormatTok->is(Kind: tok::arrow)) {
1637 FormatTok->setFinalizedType(TT_CaseLabelArrow);
1638 Default->setFinalizedType(TT_SwitchExpressionLabel);
1639 parseLabel();
1640 return;
1641 }
1642 // e.g. "default void f() {}" in a Java interface.
1643 break;
1644 }
1645 case tok::kw_case:
1646 // Proto: there are no switch/case statements.
1647 if (Style.Language == FormatStyle::LK_Proto) {
1648 nextToken();
1649 return;
1650 }
1651 if (Style.isVerilog()) {
1652 parseBlock();
1653 addUnwrappedLine();
1654 return;
1655 }
1656 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1657 // 'case: string' field declaration.
1658 nextToken();
1659 break;
1660 }
1661 parseCaseLabel();
1662 return;
1663 case tok::kw_goto:
1664 nextToken();
1665 if (FormatTok->is(Kind: tok::kw_case))
1666 nextToken();
1667 break;
1668 case tok::kw_try:
1669 case tok::kw___try:
1670 if (Style.isJavaScript() && Line->MustBeDeclaration) {
1671 // field/method declaration.
1672 break;
1673 }
1674 parseTryCatch();
1675 return;
1676 case tok::kw_extern:
1677 if (Style.isVerilog()) {
1678 // In Verilog an extern module declaration looks like a start of module.
1679 // But there is no body and endmodule. So we handle it separately.
1680 parseVerilogExtern();
1681 return;
1682 }
1683 nextToken();
1684 if (FormatTok->is(Kind: tok::string_literal)) {
1685 nextToken();
1686 if (FormatTok->is(Kind: tok::l_brace)) {
1687 if (Style.BraceWrapping.AfterExternBlock)
1688 addUnwrappedLine();
1689 // Either we indent or for backwards compatibility we follow the
1690 // AfterExternBlock style.
1691 unsigned AddLevels =
1692 (Style.IndentExternBlock == FormatStyle::IEBS_Indent) ||
1693 (Style.BraceWrapping.AfterExternBlock &&
1694 Style.IndentExternBlock ==
1695 FormatStyle::IEBS_AfterExternBlock)
1696 ? 1u
1697 : 0u;
1698 parseBlock(/*MustBeDeclaration=*/true, AddLevels);
1699 addUnwrappedLine();
1700 return;
1701 }
1702 }
1703 break;
1704 case tok::kw_export:
1705 if (IsCpp) {
1706 nextToken();
1707 if (FormatTok->is(Kind: tok::kw_namespace)) {
1708 parseNamespace();
1709 return;
1710 }
1711 if (FormatTok->is(Kind: tok::l_brace)) {
1712 parseCppExportBlock();
1713 return;
1714 }
1715 if (FormatTok->is(II: Keywords.kw_module) && parseModuleDecl())
1716 return;
1717 if (FormatTok->is(II: Keywords.kw_import) && parseImportDecl())
1718 return;
1719 break;
1720 }
1721 if (Style.isJavaScript()) {
1722 parseJavaScriptEs6ImportExport();
1723 return;
1724 }
1725 if (Style.isVerilog()) {
1726 parseVerilogExtern();
1727 return;
1728 }
1729 break;
1730 case tok::kw_inline:
1731 nextToken();
1732 if (FormatTok->is(Kind: tok::kw_namespace)) {
1733 parseNamespace();
1734 return;
1735 }
1736 break;
1737 case tok::identifier:
1738 if (FormatTok->is(TT: TT_ForEachMacro)) {
1739 parseForOrWhileLoop();
1740 return;
1741 }
1742 if (FormatTok->is(TT: TT_MacroBlockBegin)) {
1743 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
1744 /*MunchSemi=*/false);
1745 return;
1746 }
1747 if (FormatTok->is(II: Keywords.kw_import)) {
1748 if (IsCpp && parseImportDecl())
1749 return;
1750 if (Style.isJavaScript()) {
1751 parseJavaScriptEs6ImportExport();
1752 return;
1753 }
1754 if (Style.Language == FormatStyle::LK_Proto) {
1755 nextToken();
1756 if (FormatTok->is(Kind: tok::kw_public))
1757 nextToken();
1758 if (FormatTok->isNot(Kind: tok::string_literal))
1759 return;
1760 nextToken();
1761 if (FormatTok->is(Kind: tok::semi))
1762 nextToken();
1763 addUnwrappedLine();
1764 return;
1765 }
1766 if (Style.isVerilog()) {
1767 parseVerilogExtern();
1768 return;
1769 }
1770 }
1771 if (IsCpp) {
1772 if (FormatTok->is(II: Keywords.kw_module) && parseModuleDecl())
1773 return;
1774 if (FormatTok->isOneOf(K1: Keywords.kw_signals, K2: Keywords.kw_qsignals,
1775 Ks: Keywords.kw_slots, Ks: Keywords.kw_qslots)) {
1776 nextToken();
1777 if (FormatTok->is(Kind: tok::colon)) {
1778 nextToken();
1779 addUnwrappedLine();
1780 return;
1781 }
1782 }
1783 if (FormatTok->is(TT: TT_StatementMacro)) {
1784 parseStatementMacro();
1785 return;
1786 }
1787 if (FormatTok->is(TT: TT_NamespaceMacro)) {
1788 parseNamespace();
1789 return;
1790 }
1791 }
1792 // In Verilog labels can be any expression, so we don't do them here.
1793 // JS doesn't have macros, and within classes colons indicate fields, not
1794 // labels.
1795 // TableGen doesn't have labels.
1796 if (!Style.isJavaScript() && !Style.isVerilog() && !Style.isTableGen() &&
1797 Tokens->peekNextToken()->is(Kind: tok::colon) && !Line->MustBeDeclaration) {
1798 nextToken();
1799 if (!Line->InMacroBody || CurrentLines->size() > 1)
1800 Line->Tokens.begin()->Tok->MustBreakBefore = true;
1801 FormatTok->setFinalizedType(TT_GotoLabelColon);
1802 parseLabel(/*IsGotoLabel=*/true);
1803 if (HasLabel)
1804 *HasLabel = true;
1805 return;
1806 }
1807 if (Style.isJava() && FormatTok->is(II: Keywords.kw_record)) {
1808 parseRecord(/*ParseAsExpr=*/false, /*IsJavaRecord=*/true);
1809 addUnwrappedLine();
1810 return;
1811 }
1812 // In all other cases, parse the declaration.
1813 break;
1814 default:
1815 break;
1816 }
1817
1818 bool SeenEqual = false;
1819 for (const bool InRequiresExpression =
1820 OpeningBrace && OpeningBrace->isOneOf(K1: TT_RequiresExpressionLBrace,
1821 K2: TT_CompoundRequirementLBrace);
1822 !eof();) {
1823 const FormatToken *Previous = FormatTok->Previous;
1824 switch (FormatTok->Tok.getKind()) {
1825 case tok::at:
1826 nextToken();
1827 if (FormatTok->is(Kind: tok::l_brace)) {
1828 nextToken();
1829 parseBracedList();
1830 break;
1831 }
1832 if (Style.isJava() && FormatTok->is(II: Keywords.kw_interface)) {
1833 nextToken();
1834 break;
1835 }
1836 switch (bool IsAutoRelease = false; FormatTok->Tok.getObjCKeywordID()) {
1837 case tok::objc_public:
1838 case tok::objc_protected:
1839 case tok::objc_package:
1840 case tok::objc_private:
1841 return parseAccessSpecifier();
1842 case tok::objc_interface:
1843 case tok::objc_implementation:
1844 return parseObjCInterfaceOrImplementation();
1845 case tok::objc_protocol:
1846 if (parseObjCProtocol())
1847 return;
1848 break;
1849 case tok::objc_end:
1850 return; // Handled by the caller.
1851 case tok::objc_optional:
1852 case tok::objc_required:
1853 nextToken();
1854 addUnwrappedLine();
1855 return;
1856 case tok::objc_autoreleasepool:
1857 IsAutoRelease = true;
1858 [[fallthrough]];
1859 case tok::objc_synchronized:
1860 nextToken();
1861 if (!IsAutoRelease && FormatTok->is(Kind: tok::l_paren)) {
1862 // Skip synchronization object
1863 parseParens();
1864 }
1865 if (FormatTok->is(Kind: tok::l_brace)) {
1866 if (Style.BraceWrapping.AfterControlStatement ==
1867 FormatStyle::BWACS_Always) {
1868 addUnwrappedLine();
1869 }
1870 parseBlock();
1871 }
1872 addUnwrappedLine();
1873 return;
1874 case tok::objc_try:
1875 // This branch isn't strictly necessary (the kw_try case below would
1876 // do this too after the tok::at is parsed above). But be explicit.
1877 parseTryCatch();
1878 return;
1879 default:
1880 break;
1881 }
1882 break;
1883 case tok::kw_requires: {
1884 if (IsCpp) {
1885 bool ParsedClause = parseRequires(SeenEqual);
1886 if (ParsedClause)
1887 return;
1888 } else {
1889 nextToken();
1890 }
1891 break;
1892 }
1893 case tok::kw_enum:
1894 // Ignore if this is part of "template <enum ..." or "... -> enum" or
1895 // "template <..., enum ...>".
1896 if (Previous && Previous->isOneOf(K1: tok::less, K2: tok::arrow, Ks: tok::comma)) {
1897 nextToken();
1898 break;
1899 }
1900
1901 // parseEnum falls through and does not yet add an unwrapped line as an
1902 // enum definition can start a structural element.
1903 if (!parseEnum())
1904 break;
1905 // This only applies to C++ and Verilog.
1906 if (!IsCpp && !Style.isVerilog()) {
1907 addUnwrappedLine();
1908 return;
1909 }
1910 break;
1911 case tok::kw_typedef:
1912 nextToken();
1913 if (FormatTok->isOneOf(K1: Keywords.kw_NS_ENUM, K2: Keywords.kw_NS_OPTIONS,
1914 Ks: Keywords.kw_CF_ENUM, Ks: Keywords.kw_CF_OPTIONS,
1915 Ks: Keywords.kw_CF_CLOSED_ENUM,
1916 Ks: Keywords.kw_NS_CLOSED_ENUM)) {
1917 parseEnum();
1918 }
1919 break;
1920 case tok::kw_class:
1921 if (Style.isVerilog()) {
1922 parseBlock();
1923 addUnwrappedLine();
1924 return;
1925 }
1926 if (Style.isTableGen()) {
1927 // Do nothing special. In this case the l_brace becomes FunctionLBrace.
1928 // This is same as def and so on.
1929 nextToken();
1930 break;
1931 }
1932 [[fallthrough]];
1933 case tok::kw_struct:
1934 case tok::kw_union:
1935 if (parseStructLike())
1936 return;
1937 break;
1938 case tok::kw_decltype:
1939 nextToken();
1940 if (FormatTok->is(Kind: tok::l_paren)) {
1941 parseParens();
1942 if (FormatTok->Previous &&
1943 FormatTok->Previous->endsSequence(K1: tok::r_paren, Tokens: tok::kw_auto,
1944 Tokens: tok::l_paren)) {
1945 Line->SeenDecltypeAuto = true;
1946 }
1947 }
1948 break;
1949 case tok::period:
1950 nextToken();
1951 // In Java, classes have an implicit static member "class".
1952 if (Style.isJava() && FormatTok && FormatTok->is(Kind: tok::kw_class))
1953 nextToken();
1954 if (Style.isJavaScript() && FormatTok &&
1955 FormatTok->Tok.getIdentifierInfo()) {
1956 // JavaScript only has pseudo keywords, all keywords are allowed to
1957 // appear in "IdentifierName" positions. See http://es5.github.io/#x7.6
1958 nextToken();
1959 }
1960 break;
1961 case tok::semi:
1962 nextToken();
1963 addUnwrappedLine();
1964 return;
1965 case tok::r_brace:
1966 addUnwrappedLine();
1967 return;
1968 case tok::string_literal:
1969 if (Style.isVerilog() && FormatTok->is(TT: TT_VerilogProtected)) {
1970 FormatTok->Finalized = true;
1971 nextToken();
1972 addUnwrappedLine();
1973 return;
1974 }
1975 nextToken();
1976 break;
1977 case tok::l_paren: {
1978 parseParens();
1979 // Break the unwrapped line if a K&R C function definition has a parameter
1980 // declaration.
1981 if (OpeningBrace || !IsCpp || !Previous || eof())
1982 break;
1983 if (isC78ParameterDecl(Tok: FormatTok,
1984 Next: Tokens->peekNextToken(/*SkipComment=*/true),
1985 FuncName: Previous)) {
1986 addUnwrappedLine();
1987 return;
1988 }
1989 break;
1990 }
1991 case tok::kw_operator:
1992 nextToken();
1993 if (FormatTok->isBinaryOperator())
1994 nextToken();
1995 break;
1996 case tok::caret: {
1997 const auto *Prev = FormatTok->getPreviousNonComment();
1998 nextToken();
1999 if (Prev && Prev->is(Kind: tok::identifier))
2000 break;
2001 // Block return type.
2002 if (FormatTok->Tok.isAnyIdentifier() || FormatTok->isTypeName(LangOpts)) {
2003 nextToken();
2004 // Return types: ObjC generics and protocol qualifiers are ok too.
2005 if (FormatTok->is(Kind: tok::less)) {
2006 nextToken();
2007 parseBracedList(/*IsAngleBracket=*/true);
2008 }
2009 // Return types: pointers are ok too.
2010 while (FormatTok->is(Kind: tok::star))
2011 nextToken();
2012 }
2013 // Block argument list.
2014 if (FormatTok->is(Kind: tok::l_paren))
2015 parseParens();
2016 // Block body.
2017 if (FormatTok->is(Kind: tok::l_brace))
2018 parseChildBlock();
2019 break;
2020 }
2021 case tok::l_brace:
2022 if (InRequiresExpression)
2023 FormatTok->setFinalizedType(TT_BracedListLBrace);
2024 if (!tryToParsePropertyAccessor() && !tryToParseBracedList()) {
2025 IsDecltypeAutoFunction = Line->SeenDecltypeAuto;
2026 // A block outside of parentheses must be the last part of a
2027 // structural element.
2028 // FIXME: Figure out cases where this is not true, and add projections
2029 // for them (the one we know is missing are lambdas).
2030 if (Style.isJava() &&
2031 Line->Tokens.front().Tok->is(II: Keywords.kw_synchronized)) {
2032 // If necessary, we could set the type to something different than
2033 // TT_FunctionLBrace.
2034 if (Style.BraceWrapping.AfterControlStatement ==
2035 FormatStyle::BWACS_Always) {
2036 addUnwrappedLine();
2037 }
2038 } else if (Style.BraceWrapping.AfterFunction) {
2039 addUnwrappedLine();
2040 }
2041 if (!Previous || Previous->isNot(Kind: TT_TypeDeclarationParen))
2042 FormatTok->setFinalizedType(TT_FunctionLBrace);
2043 parseBlock();
2044 IsDecltypeAutoFunction = false;
2045 addUnwrappedLine();
2046 return;
2047 }
2048 // Otherwise this was a braced init list, and the structural
2049 // element continues.
2050 break;
2051 case tok::kw_try:
2052 if (Style.isJavaScript() && Line->MustBeDeclaration) {
2053 // field/method declaration.
2054 nextToken();
2055 break;
2056 }
2057 // We arrive here when parsing function-try blocks.
2058 if (Style.BraceWrapping.AfterFunction)
2059 addUnwrappedLine();
2060 parseTryCatch();
2061 return;
2062 case tok::identifier: {
2063 if (Style.isCSharp() && FormatTok->is(II: Keywords.kw_where) &&
2064 Line->MustBeDeclaration) {
2065 addUnwrappedLine();
2066 parseCSharpGenericTypeConstraint();
2067 break;
2068 }
2069 if (FormatTok->is(TT: TT_MacroBlockEnd)) {
2070 addUnwrappedLine();
2071 return;
2072 }
2073
2074 // Function declarations (as opposed to function expressions) are parsed
2075 // on their own unwrapped line by continuing this loop. Function
2076 // expressions (functions that are not on their own line) must not create
2077 // a new unwrapped line, so they are special cased below.
2078 size_t TokenCount = Line->Tokens.size();
2079 if (Style.isJavaScript() && FormatTok->is(II: Keywords.kw_function) &&
2080 (TokenCount > 1 ||
2081 (TokenCount == 1 &&
2082 Line->Tokens.front().Tok->isNot(Kind: Keywords.kw_async)))) {
2083 tryToParseJSFunction();
2084 break;
2085 }
2086 if ((Style.isJavaScript() || Style.isJava()) &&
2087 FormatTok->is(II: Keywords.kw_interface)) {
2088 if (Style.isJavaScript()) {
2089 // In JavaScript/TypeScript, "interface" can be used as a standalone
2090 // identifier, e.g. in `var interface = 1;`. If "interface" is
2091 // followed by another identifier, it is very like to be an actual
2092 // interface declaration.
2093 unsigned StoredPosition = Tokens->getPosition();
2094 FormatToken *Next = Tokens->getNextToken();
2095 FormatTok = Tokens->setPosition(StoredPosition);
2096 if (!mustBeJSIdent(Keywords, FormatTok: Next)) {
2097 nextToken();
2098 break;
2099 }
2100 }
2101 parseRecord();
2102 addUnwrappedLine();
2103 return;
2104 }
2105
2106 if (Style.isVerilog()) {
2107 if (FormatTok->is(II: Keywords.kw_table)) {
2108 parseVerilogTable();
2109 return;
2110 }
2111 if (Keywords.isVerilogBegin(Tok: *FormatTok) ||
2112 Keywords.isVerilogHierarchy(Tok: *FormatTok)) {
2113 parseBlock();
2114 addUnwrappedLine();
2115 return;
2116 }
2117 }
2118
2119 if (!IsCpp && FormatTok->is(II: Keywords.kw_interface)) {
2120 if (parseStructLike())
2121 return;
2122 break;
2123 }
2124
2125 if (IsCpp && FormatTok->is(TT: TT_StatementMacro)) {
2126 parseStatementMacro();
2127 return;
2128 }
2129
2130 // See if the following token should start a new unwrapped line.
2131 StringRef Text = FormatTok->TokenText;
2132
2133 FormatToken *PreviousToken = FormatTok;
2134 nextToken();
2135
2136 // JS doesn't have macros, and within classes colons indicate fields, not
2137 // labels.
2138 if (Style.isJavaScript())
2139 break;
2140
2141 auto OneTokenSoFar = [&]() {
2142 auto I = Line->Tokens.begin(), E = Line->Tokens.end();
2143 while (I != E && I->Tok->is(Kind: tok::comment))
2144 ++I;
2145 if (Style.isVerilog())
2146 while (I != E && I->Tok->is(Kind: tok::hash))
2147 ++I;
2148 return I != E && (++I == E);
2149 };
2150 if (OneTokenSoFar()) {
2151 // Recognize function-like macro usages without trailing semicolon as
2152 // well as free-standing macros like Q_OBJECT.
2153 bool FunctionLike = FormatTok->is(Kind: tok::l_paren);
2154 if (FunctionLike)
2155 parseParens();
2156
2157 bool FollowedByNewline =
2158 CommentsBeforeNextToken.empty()
2159 ? FormatTok->NewlinesBefore > 0
2160 : CommentsBeforeNextToken.front()->NewlinesBefore > 0;
2161
2162 if (FollowedByNewline &&
2163 (Text.size() >= 5 ||
2164 (FunctionLike && FormatTok->isNot(Kind: tok::l_paren))) &&
2165 tokenCanStartNewLine(Tok: *FormatTok) && Text == Text.upper()) {
2166 if (PreviousToken->isNot(Kind: TT_UntouchableMacroFunc))
2167 PreviousToken->setFinalizedType(TT_FunctionLikeOrFreestandingMacro);
2168 addUnwrappedLine();
2169 return;
2170 }
2171 }
2172 break;
2173 }
2174 case tok::equal:
2175 if ((Style.isJavaScript() || Style.isCSharp()) &&
2176 FormatTok->is(TT: TT_FatArrow)) {
2177 tryToParseChildBlock();
2178 break;
2179 }
2180
2181 SeenEqual = true;
2182 nextToken();
2183 if (FormatTok->is(Kind: tok::l_brace)) {
2184 // C# needs this change to ensure that array initialisers and object
2185 // initialisers are indented the same way. In TypeScript, the brace
2186 // can also be an object type definition.
2187 if (!Style.isJavaScript())
2188 FormatTok->setBlockKind(BK_BracedInit);
2189 // TableGen's defset statement has syntax of the form,
2190 // `defset <type> <name> = { <statement>... }`
2191 if (Style.isTableGen() &&
2192 Line->Tokens.begin()->Tok->is(II: Keywords.kw_defset)) {
2193 FormatTok->setFinalizedType(TT_FunctionLBrace);
2194 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
2195 /*MunchSemi=*/false);
2196 addUnwrappedLine();
2197 break;
2198 }
2199 nextToken();
2200 parseBracedList();
2201 } else if (Style.Language == FormatStyle::LK_Proto &&
2202 FormatTok->is(Kind: tok::less)) {
2203 nextToken();
2204 parseBracedList(/*IsAngleBracket=*/true);
2205 }
2206 break;
2207 case tok::l_square:
2208 parseSquare();
2209 break;
2210 case tok::kw_new:
2211 if (Style.isCSharp() &&
2212 (Tokens->peekNextToken()->isAccessSpecifierKeyword() ||
2213 (Previous && Previous->isAccessSpecifierKeyword()))) {
2214 nextToken();
2215 } else {
2216 parseNew();
2217 }
2218 break;
2219 case tok::kw_switch:
2220 if (Style.isJava())
2221 parseSwitch(/*IsExpr=*/true);
2222 else
2223 nextToken();
2224 break;
2225 case tok::kw_case:
2226 // Proto: there are no switch/case statements.
2227 if (Style.Language == FormatStyle::LK_Proto) {
2228 nextToken();
2229 return;
2230 }
2231 // In Verilog switch is called case.
2232 if (Style.isVerilog()) {
2233 parseBlock();
2234 addUnwrappedLine();
2235 return;
2236 }
2237 if (Style.isJavaScript() && Line->MustBeDeclaration) {
2238 // 'case: string' field declaration.
2239 nextToken();
2240 break;
2241 }
2242 parseCaseLabel();
2243 break;
2244 case tok::kw_default:
2245 nextToken();
2246 if (Style.isVerilog()) {
2247 if (FormatTok->is(Kind: tok::colon)) {
2248 // The label will be handled in the next iteration.
2249 break;
2250 }
2251 if (FormatTok->is(II: Keywords.kw_clocking)) {
2252 // A default clocking block.
2253 parseBlock();
2254 addUnwrappedLine();
2255 return;
2256 }
2257 parseVerilogCaseLabel();
2258 return;
2259 }
2260 break;
2261 case tok::colon:
2262 nextToken();
2263 if (Style.isVerilog()) {
2264 parseVerilogCaseLabel();
2265 return;
2266 }
2267 break;
2268 case tok::greater:
2269 nextToken();
2270 if (FormatTok->is(Kind: tok::l_brace))
2271 FormatTok->Previous->setFinalizedType(TT_TemplateCloser);
2272 break;
2273 default:
2274 nextToken();
2275 break;
2276 }
2277 }
2278}
2279
2280bool UnwrappedLineParser::tryToParsePropertyAccessor() {
2281 assert(FormatTok->is(tok::l_brace));
2282 if (!Style.isCSharp())
2283 return false;
2284 // See if it's a property accessor.
2285 if (!FormatTok->Previous || FormatTok->Previous->isNot(Kind: tok::identifier))
2286 return false;
2287
2288 // See if we are inside a property accessor.
2289 //
2290 // Record the current tokenPosition so that we can advance and
2291 // reset the current token. `Next` is not set yet so we need
2292 // another way to advance along the token stream.
2293 unsigned int StoredPosition = Tokens->getPosition();
2294 FormatToken *Tok = Tokens->getNextToken();
2295
2296 // A trivial property accessor is of the form:
2297 // { [ACCESS_SPECIFIER] [get]; [ACCESS_SPECIFIER] [set|init] }
2298 // Track these as they do not require line breaks to be introduced.
2299 bool HasSpecialAccessor = false;
2300 bool IsTrivialPropertyAccessor = true;
2301 bool HasAttribute = false;
2302 while (!eof()) {
2303 if (const bool IsAccessorKeyword =
2304 Tok->isOneOf(K1: Keywords.kw_get, K2: Keywords.kw_init, Ks: Keywords.kw_set);
2305 IsAccessorKeyword || Tok->isAccessSpecifierKeyword() ||
2306 Tok->isOneOf(K1: tok::l_square, K2: tok::semi, Ks: Keywords.kw_internal)) {
2307 if (IsAccessorKeyword)
2308 HasSpecialAccessor = true;
2309 else if (Tok->is(Kind: tok::l_square))
2310 HasAttribute = true;
2311 Tok = Tokens->getNextToken();
2312 continue;
2313 }
2314 if (Tok->isNot(Kind: tok::r_brace))
2315 IsTrivialPropertyAccessor = false;
2316 break;
2317 }
2318
2319 if (!HasSpecialAccessor || HasAttribute) {
2320 Tokens->setPosition(StoredPosition);
2321 return false;
2322 }
2323
2324 // Try to parse the property accessor:
2325 // https://docs.microsoft.com/en-us/dotnet/csharp/programming-guide/classes-and-structs/properties
2326 Tokens->setPosition(StoredPosition);
2327 if (!IsTrivialPropertyAccessor && Style.BraceWrapping.AfterFunction)
2328 addUnwrappedLine();
2329 nextToken();
2330 do {
2331 switch (FormatTok->Tok.getKind()) {
2332 case tok::r_brace:
2333 nextToken();
2334 if (FormatTok->is(Kind: tok::equal)) {
2335 while (!eof() && FormatTok->isNot(Kind: tok::semi))
2336 nextToken();
2337 nextToken();
2338 }
2339 addUnwrappedLine();
2340 return true;
2341 case tok::l_brace:
2342 ++Line->Level;
2343 parseBlock(/*MustBeDeclaration=*/true);
2344 addUnwrappedLine();
2345 --Line->Level;
2346 break;
2347 case tok::equal:
2348 if (FormatTok->is(TT: TT_FatArrow)) {
2349 ++Line->Level;
2350 do {
2351 nextToken();
2352 } while (!eof() && FormatTok->isNot(Kind: tok::semi));
2353 nextToken();
2354 addUnwrappedLine();
2355 --Line->Level;
2356 break;
2357 }
2358 nextToken();
2359 break;
2360 default:
2361 if (FormatTok->isOneOf(K1: Keywords.kw_get, K2: Keywords.kw_init,
2362 Ks: Keywords.kw_set) &&
2363 !IsTrivialPropertyAccessor) {
2364 // Non-trivial get/set needs to be on its own line.
2365 addUnwrappedLine();
2366 }
2367 nextToken();
2368 }
2369 } while (!eof());
2370
2371 // Unreachable for well-formed code (paired '{' and '}').
2372 return true;
2373}
2374
2375bool UnwrappedLineParser::tryToParseLambda() {
2376 assert(FormatTok->is(tok::l_square));
2377 if (!IsCpp) {
2378 nextToken();
2379 return false;
2380 }
2381 FormatToken &LSquare = *FormatTok;
2382 if (!tryToParseLambdaIntroducer())
2383 return false;
2384
2385 FormatToken *Arrow = nullptr;
2386 bool InTemplateParameterList = false;
2387
2388 while (FormatTok->isNot(Kind: tok::l_brace)) {
2389 if (FormatTok->isTypeName(LangOpts) || FormatTok->isAttribute()) {
2390 nextToken();
2391 continue;
2392 }
2393 switch (FormatTok->Tok.getKind()) {
2394 case tok::l_brace:
2395 break;
2396 case tok::l_paren:
2397 parseParens(/*AmpAmpTokenType=*/StarAndAmpTokenType: TT_PointerOrReference);
2398 break;
2399 case tok::l_square:
2400 parseSquare();
2401 break;
2402 case tok::less:
2403 assert(FormatTok->Previous);
2404 if (FormatTok->Previous->is(Kind: tok::r_square))
2405 InTemplateParameterList = true;
2406 nextToken();
2407 break;
2408 case tok::kw_auto:
2409 case tok::kw_class:
2410 case tok::kw_struct:
2411 case tok::kw_union:
2412 case tok::kw_template:
2413 case tok::kw_typename:
2414 case tok::amp:
2415 case tok::star:
2416 case tok::kw_const:
2417 case tok::kw_constexpr:
2418 case tok::kw_consteval:
2419 case tok::comma:
2420 case tok::greater:
2421 case tok::identifier:
2422 case tok::numeric_constant:
2423 case tok::coloncolon:
2424 case tok::kw_mutable:
2425 case tok::kw_noexcept:
2426 case tok::kw_static:
2427 nextToken();
2428 break;
2429 // Specialization of a template with an integer parameter can contain
2430 // arithmetic, logical, comparison and ternary operators.
2431 //
2432 // FIXME: This also accepts sequences of operators that are not in the scope
2433 // of a template argument list.
2434 //
2435 // In a C++ lambda a template type can only occur after an arrow. We use
2436 // this as an heuristic to distinguish between Objective-C expressions
2437 // followed by an `a->b` expression, such as:
2438 // ([obj func:arg] + a->b)
2439 // Otherwise the code below would parse as a lambda.
2440 case tok::plus:
2441 case tok::minus:
2442 case tok::exclaim:
2443 case tok::tilde:
2444 case tok::slash:
2445 case tok::percent:
2446 case tok::lessless:
2447 case tok::pipe:
2448 case tok::pipepipe:
2449 case tok::ampamp:
2450 case tok::caret:
2451 case tok::equalequal:
2452 case tok::exclaimequal:
2453 case tok::greaterequal:
2454 case tok::lessequal:
2455 case tok::question:
2456 case tok::colon:
2457 case tok::ellipsis:
2458 case tok::kw_true:
2459 case tok::kw_false:
2460 if (Arrow || InTemplateParameterList) {
2461 nextToken();
2462 break;
2463 }
2464 return true;
2465 case tok::arrow:
2466 Arrow = FormatTok;
2467 nextToken();
2468 break;
2469 case tok::kw_requires:
2470 parseRequiresClause();
2471 break;
2472 case tok::equal:
2473 if (!InTemplateParameterList)
2474 return true;
2475 nextToken();
2476 break;
2477 default:
2478 return true;
2479 }
2480 }
2481
2482 FormatTok->setFinalizedType(TT_LambdaLBrace);
2483 LSquare.setFinalizedType(TT_LambdaLSquare);
2484
2485 if (Arrow)
2486 Arrow->setFinalizedType(TT_LambdaArrow);
2487
2488 NestedLambdas.push_back(Elt: Line->SeenDecltypeAuto);
2489 parseChildBlock();
2490 assert(!NestedLambdas.empty());
2491 NestedLambdas.pop_back();
2492
2493 return true;
2494}
2495
2496bool UnwrappedLineParser::tryToParseLambdaIntroducer() {
2497 const FormatToken *Previous = FormatTok->Previous;
2498 const FormatToken *LeftSquare = FormatTok;
2499 nextToken();
2500 if (Previous) {
2501 const auto *PrevPrev = Previous->getPreviousNonComment();
2502 if (Previous->is(Kind: tok::star) && PrevPrev && PrevPrev->isTypeName(LangOpts))
2503 return false;
2504 if (Previous->closesScope()) {
2505 // Not a potential C-style cast.
2506 if (Previous->isNot(Kind: tok::r_paren))
2507 return false;
2508 // Lambdas can be cast to function types only, e.g. `std::function<int()>`
2509 // and `int (*)()`.
2510 if (!PrevPrev || PrevPrev->isNoneOf(Ks: tok::greater, Ks: tok::r_paren))
2511 return false;
2512 }
2513 if (Previous && Previous->Tok.getIdentifierInfo() &&
2514 Previous->isNoneOf(Ks: tok::kw_return, Ks: tok::kw_co_await, Ks: tok::kw_co_yield,
2515 Ks: tok::kw_co_return)) {
2516 return false;
2517 }
2518 }
2519 if (LeftSquare->isCppStructuredBinding(IsCpp))
2520 return false;
2521 if (FormatTok->is(Kind: tok::l_square) || tok::isLiteral(K: FormatTok->Tok.getKind()))
2522 return false;
2523 if (FormatTok->is(Kind: tok::r_square)) {
2524 const FormatToken *Next = Tokens->peekNextToken(/*SkipComment=*/true);
2525 if (Next->is(Kind: tok::greater))
2526 return false;
2527 }
2528 parseSquare(/*LambdaIntroducer=*/true);
2529 return true;
2530}
2531
2532void UnwrappedLineParser::tryToParseJSFunction() {
2533 assert(FormatTok->is(Keywords.kw_function));
2534 if (FormatTok->is(II: Keywords.kw_async))
2535 nextToken();
2536 // Consume "function".
2537 nextToken();
2538
2539 // Consume * (generator function). Treat it like C++'s overloaded operators.
2540 if (FormatTok->is(Kind: tok::star)) {
2541 FormatTok->setFinalizedType(TT_OverloadedOperator);
2542 nextToken();
2543 }
2544
2545 // Consume function name.
2546 if (FormatTok->is(Kind: tok::identifier))
2547 nextToken();
2548
2549 if (FormatTok->isNot(Kind: tok::l_paren))
2550 return;
2551
2552 // Parse formal parameter list.
2553 parseParens();
2554
2555 if (FormatTok->is(Kind: tok::colon)) {
2556 // Parse a type definition.
2557 nextToken();
2558
2559 // Eat the type declaration. For braced inline object types, balance braces,
2560 // otherwise just parse until finding an l_brace for the function body.
2561 if (FormatTok->is(Kind: tok::l_brace))
2562 tryToParseBracedList();
2563 else
2564 while (FormatTok->isNoneOf(Ks: tok::l_brace, Ks: tok::semi) && !eof())
2565 nextToken();
2566 }
2567
2568 if (FormatTok->is(Kind: tok::semi))
2569 return;
2570
2571 parseChildBlock();
2572}
2573
2574bool UnwrappedLineParser::tryToParseBracedList() {
2575 if (FormatTok->is(BBK: BK_Unknown))
2576 calculateBraceTypes();
2577 assert(FormatTok->isNot(BK_Unknown));
2578 if (FormatTok->is(BBK: BK_Block))
2579 return false;
2580 nextToken();
2581 parseBracedList();
2582 return true;
2583}
2584
2585bool UnwrappedLineParser::tryToParseChildBlock() {
2586 assert(Style.isJavaScript() || Style.isCSharp());
2587 assert(FormatTok->is(TT_FatArrow));
2588 // Fat arrows (=>) have tok::TokenKind tok::equal but TokenType TT_FatArrow.
2589 // They always start an expression or a child block if followed by a curly
2590 // brace.
2591 nextToken();
2592 if (FormatTok->isNot(Kind: tok::l_brace))
2593 return false;
2594 parseChildBlock();
2595 return true;
2596}
2597
2598bool UnwrappedLineParser::parseBracedList(bool IsAngleBracket, bool IsEnum) {
2599 assert(!IsAngleBracket || !IsEnum);
2600 bool HasError = false;
2601
2602 // FIXME: Once we have an expression parser in the UnwrappedLineParser,
2603 // replace this by using parseAssignmentExpression() inside.
2604 do {
2605 if (Style.isCSharp() && FormatTok->is(TT: TT_FatArrow) &&
2606 tryToParseChildBlock()) {
2607 continue;
2608 }
2609 if (Style.isJavaScript()) {
2610 if (FormatTok->is(II: Keywords.kw_function)) {
2611 tryToParseJSFunction();
2612 continue;
2613 }
2614 if (FormatTok->is(Kind: tok::l_brace)) {
2615 // Could be a method inside of a braced list `{a() { return 1; }}`.
2616 if (tryToParseBracedList())
2617 continue;
2618 parseChildBlock();
2619 }
2620 }
2621 if (FormatTok->is(Kind: IsAngleBracket ? tok::greater : tok::r_brace)) {
2622 if (IsEnum) {
2623 FormatTok->setBlockKind(BK_Block);
2624 if (!Style.AllowShortEnumsOnASingleLine)
2625 addUnwrappedLine();
2626 }
2627 nextToken();
2628 return !HasError;
2629 }
2630 switch (FormatTok->Tok.getKind()) {
2631 case tok::l_square:
2632 if (Style.isCSharp())
2633 parseSquare();
2634 else
2635 tryToParseLambda();
2636 break;
2637 case tok::l_paren:
2638 parseParens();
2639 // JavaScript can just have free standing methods and getters/setters in
2640 // object literals. Detect them by a "{" following ")".
2641 if (Style.isJavaScript()) {
2642 if (FormatTok->is(Kind: tok::l_brace))
2643 parseChildBlock();
2644 break;
2645 }
2646 break;
2647 case tok::l_brace:
2648 // Assume there are no blocks inside a braced init list apart
2649 // from the ones we explicitly parse out (like lambdas).
2650 FormatTok->setBlockKind(BK_BracedInit);
2651 if (!IsAngleBracket) {
2652 auto *Prev = FormatTok->Previous;
2653 if (Prev && Prev->is(Kind: tok::greater))
2654 Prev->setFinalizedType(TT_TemplateCloser);
2655 }
2656 nextToken();
2657 parseBracedList();
2658 break;
2659 case tok::less:
2660 nextToken();
2661 if (IsAngleBracket)
2662 parseBracedList(/*IsAngleBracket=*/true);
2663 break;
2664 case tok::semi:
2665 // JavaScript (or more precisely TypeScript) can have semicolons in braced
2666 // lists (in so-called TypeMemberLists). Thus, the semicolon cannot be
2667 // used for error recovery if we have otherwise determined that this is
2668 // a braced list.
2669 if (Style.isJavaScript()) {
2670 nextToken();
2671 break;
2672 }
2673 HasError = true;
2674 if (!IsEnum)
2675 return false;
2676 nextToken();
2677 break;
2678 case tok::comma:
2679 nextToken();
2680 if (IsEnum && !Style.AllowShortEnumsOnASingleLine)
2681 addUnwrappedLine();
2682 break;
2683 case tok::kw_requires:
2684 parseRequiresExpression();
2685 break;
2686 default:
2687 nextToken();
2688 break;
2689 }
2690 } while (!eof());
2691 return false;
2692}
2693
2694/// Parses a pair of parentheses (and everything between them).
2695/// \param StarAndAmpTokenType If different than TT_Unknown sets this type for
2696/// all (double) ampersands and stars. This applies for all nested scopes as
2697/// well, this is disabled within a (potential) template argument <>, and thus
2698/// also if we find only a <.
2699///
2700/// Returns whether there is a `=` token between the parentheses.
2701bool UnwrappedLineParser::parseParens(TokenType StarAndAmpTokenType,
2702 bool InMacroCall) {
2703 assert(FormatTok->is(tok::l_paren) && "'(' expected.");
2704 auto *LParen = FormatTok;
2705 auto *Prev = FormatTok->Previous;
2706 bool SeenComma = false;
2707 bool SeenEqual = false;
2708 bool MightBeFoldExpr = false;
2709 unsigned ExcessLess = 0;
2710 nextToken();
2711 const bool MightBeStmtExpr = FormatTok->is(Kind: tok::l_brace);
2712 if (!InMacroCall && Prev && Prev->is(TT: TT_FunctionLikeMacro))
2713 InMacroCall = true;
2714 do {
2715 switch (FormatTok->Tok.getKind()) {
2716 case tok::l_paren:
2717 if (parseParens(StarAndAmpTokenType: ExcessLess == 0 ? StarAndAmpTokenType : TT_Unknown,
2718 InMacroCall)) {
2719 SeenEqual = true;
2720 }
2721 if (Style.isJava() && FormatTok->is(Kind: tok::l_brace))
2722 parseChildBlock();
2723 break;
2724 case tok::r_paren: {
2725 auto *RParen = FormatTok;
2726 nextToken();
2727 if (Prev) {
2728 auto OptionalParens = [&] {
2729 if (Style.RemoveParentheses == FormatStyle::RPS_Leave ||
2730 MightBeStmtExpr || MightBeFoldExpr || SeenComma || InMacroCall ||
2731 Line->InMacroBody || RParen->getPreviousNonComment() == LParen) {
2732 return false;
2733 }
2734 const bool DoubleParens =
2735 Prev->is(Kind: tok::l_paren) && FormatTok->is(Kind: tok::r_paren);
2736 if (DoubleParens) {
2737 const auto *PrevPrev = Prev->getPreviousNonComment();
2738 const bool Excluded =
2739 PrevPrev &&
2740 (PrevPrev->isOneOf(K1: tok::kw___attribute, K2: tok::kw_decltype) ||
2741 (SeenEqual &&
2742 (PrevPrev->isOneOf(K1: tok::kw_if, K2: tok::kw_while) ||
2743 PrevPrev->endsSequence(K1: tok::kw_constexpr, Tokens: tok::kw_if))));
2744 if (!Excluded)
2745 return true;
2746 } else {
2747 const bool CommaSeparated =
2748 Prev->isOneOf(K1: tok::l_paren, K2: tok::comma) &&
2749 FormatTok->isOneOf(K1: tok::comma, K2: tok::r_paren);
2750 if (CommaSeparated &&
2751 // LParen is not preceded by ellipsis, comma.
2752 !Prev->endsSequence(K1: tok::comma, Tokens: tok::ellipsis) &&
2753 // RParen is not followed by comma, ellipsis.
2754 !(FormatTok->is(Kind: tok::comma) &&
2755 Tokens->peekNextToken()->is(Kind: tok::ellipsis))) {
2756 return true;
2757 }
2758 const bool ReturnParens =
2759 Style.RemoveParentheses == FormatStyle::RPS_ReturnStatement &&
2760 ((NestedLambdas.empty() && !IsDecltypeAutoFunction) ||
2761 (!NestedLambdas.empty() && !NestedLambdas.back())) &&
2762 Prev->isOneOf(K1: tok::kw_return, K2: tok::kw_co_return) &&
2763 FormatTok->is(Kind: tok::semi);
2764 if (ReturnParens)
2765 return true;
2766 }
2767 return false;
2768 };
2769 if (OptionalParens()) {
2770 LParen->Optional = true;
2771 RParen->Optional = true;
2772 } else if (Prev->is(TT: TT_TypenameMacro)) {
2773 LParen->setFinalizedType(TT_TypeDeclarationParen);
2774 RParen->setFinalizedType(TT_TypeDeclarationParen);
2775 } else if (Prev->is(Kind: tok::greater) && RParen->Previous == LParen) {
2776 Prev->setFinalizedType(TT_TemplateCloser);
2777 } else if (FormatTok->is(Kind: tok::l_brace) && Prev->is(Kind: tok::amp) &&
2778 !Prev->Previous) {
2779 FormatTok->setBlockKind(BK_BracedInit);
2780 }
2781 }
2782 return SeenEqual;
2783 }
2784 case tok::r_brace:
2785 // A "}" inside parenthesis is an error if there wasn't a matching "{".
2786 return SeenEqual;
2787 case tok::l_square:
2788 tryToParseLambda();
2789 break;
2790 case tok::l_brace:
2791 if (!tryToParseBracedList())
2792 parseChildBlock();
2793 break;
2794 case tok::at:
2795 nextToken();
2796 if (FormatTok->is(Kind: tok::l_brace)) {
2797 nextToken();
2798 parseBracedList();
2799 }
2800 break;
2801 case tok::comma:
2802 SeenComma = true;
2803 nextToken();
2804 break;
2805 case tok::ellipsis:
2806 MightBeFoldExpr = true;
2807 nextToken();
2808 break;
2809 case tok::equal:
2810 SeenEqual = true;
2811 if (Style.isCSharp() && FormatTok->is(TT: TT_FatArrow))
2812 tryToParseChildBlock();
2813 else
2814 nextToken();
2815 break;
2816 case tok::kw_class:
2817 if (Style.isJavaScript())
2818 parseRecord(/*ParseAsExpr=*/true);
2819 else
2820 nextToken();
2821 break;
2822 case tok::identifier:
2823 if (Style.isJavaScript() && (FormatTok->is(II: Keywords.kw_function)))
2824 tryToParseJSFunction();
2825 else
2826 nextToken();
2827 break;
2828 case tok::kw_switch:
2829 if (Style.isJava())
2830 parseSwitch(/*IsExpr=*/true);
2831 else
2832 nextToken();
2833 break;
2834 case tok::kw_requires:
2835 parseRequiresExpression();
2836 break;
2837 case tok::less:
2838 // We have here no clue whether this is a less, or a template opener, opt
2839 // out of the predefined StarAndAmpTokenType.
2840 ++ExcessLess;
2841 nextToken();
2842 break;
2843 case tok::greater:
2844 if (ExcessLess > 0)
2845 --ExcessLess;
2846 nextToken();
2847 break;
2848 case tok::star:
2849 case tok::amp:
2850 case tok::ampamp:
2851 if (StarAndAmpTokenType != TT_Unknown && ExcessLess == 0)
2852 FormatTok->setFinalizedType(StarAndAmpTokenType);
2853 [[fallthrough]];
2854 default:
2855 nextToken();
2856 break;
2857 }
2858 } while (!eof());
2859 return SeenEqual;
2860}
2861
2862void UnwrappedLineParser::parseSquare(bool LambdaIntroducer) {
2863 if (!LambdaIntroducer) {
2864 assert(FormatTok->is(tok::l_square) && "'[' expected.");
2865 if (tryToParseLambda())
2866 return;
2867 }
2868 do {
2869 switch (FormatTok->Tok.getKind()) {
2870 case tok::l_paren:
2871 parseParens();
2872 break;
2873 case tok::r_square:
2874 nextToken();
2875 return;
2876 case tok::r_brace:
2877 // A "}" inside parenthesis is an error if there wasn't a matching "{".
2878 return;
2879 case tok::l_square:
2880 parseSquare();
2881 break;
2882 case tok::l_brace: {
2883 if (!tryToParseBracedList())
2884 parseChildBlock();
2885 break;
2886 }
2887 case tok::at:
2888 case tok::colon:
2889 nextToken();
2890 if (FormatTok->is(Kind: tok::l_brace)) {
2891 nextToken();
2892 parseBracedList();
2893 }
2894 break;
2895 default:
2896 nextToken();
2897 break;
2898 }
2899 } while (!eof());
2900}
2901
2902void UnwrappedLineParser::keepAncestorBraces() {
2903 if (!Style.RemoveBracesLLVM)
2904 return;
2905
2906 const int MaxNestingLevels = 2;
2907 const int Size = NestedTooDeep.size();
2908 if (Size >= MaxNestingLevels)
2909 NestedTooDeep[Size - MaxNestingLevels] = true;
2910 NestedTooDeep.push_back(Elt: false);
2911}
2912
2913static FormatToken *getLastNonComment(const UnwrappedLine &Line) {
2914 for (const auto &Token : llvm::reverse(C: Line.Tokens))
2915 if (Token.Tok->isNot(Kind: tok::comment))
2916 return Token.Tok;
2917
2918 return nullptr;
2919}
2920
2921void UnwrappedLineParser::parseUnbracedBody(bool CheckEOF) {
2922 FormatToken *Tok = nullptr;
2923
2924 if (Style.InsertBraces && !Line->InPPDirective && !Line->Tokens.empty() &&
2925 PreprocessorDirectives.empty() && FormatTok->isNot(Kind: tok::semi)) {
2926 Tok = Style.BraceWrapping.AfterControlStatement == FormatStyle::BWACS_Never
2927 ? getLastNonComment(Line: *Line)
2928 : Line->Tokens.back().Tok;
2929 assert(Tok);
2930 if (Tok->BraceCount < 0) {
2931 assert(Tok->BraceCount == -1);
2932 Tok = nullptr;
2933 } else {
2934 Tok->BraceCount = -1;
2935 }
2936 }
2937
2938 addUnwrappedLine();
2939 ++Line->Level;
2940 ++Line->UnbracedBodyLevel;
2941 parseStructuralElement();
2942 --Line->UnbracedBodyLevel;
2943
2944 if (Tok) {
2945 assert(!Line->InPPDirective);
2946 Tok = nullptr;
2947 for (const auto &L : llvm::reverse(C&: *CurrentLines)) {
2948 if (!L.InPPDirective && getLastNonComment(Line: L)) {
2949 Tok = L.Tokens.back().Tok;
2950 break;
2951 }
2952 }
2953 assert(Tok);
2954 ++Tok->BraceCount;
2955 }
2956
2957 if (CheckEOF && eof())
2958 addUnwrappedLine();
2959
2960 --Line->Level;
2961}
2962
2963static void markOptionalBraces(FormatToken *LeftBrace) {
2964 if (!LeftBrace)
2965 return;
2966
2967 assert(LeftBrace->is(tok::l_brace));
2968
2969 FormatToken *RightBrace = LeftBrace->MatchingParen;
2970 if (!RightBrace) {
2971 assert(!LeftBrace->Optional);
2972 return;
2973 }
2974
2975 assert(RightBrace->is(tok::r_brace));
2976 assert(RightBrace->MatchingParen == LeftBrace);
2977 assert(LeftBrace->Optional == RightBrace->Optional);
2978
2979 LeftBrace->Optional = true;
2980 RightBrace->Optional = true;
2981}
2982
2983void UnwrappedLineParser::handleAttributes() {
2984 // Handle AttributeMacro, e.g. `if (x) UNLIKELY`.
2985 if (FormatTok->isAttribute())
2986 nextToken();
2987 else if (FormatTok->is(Kind: tok::l_square))
2988 handleCppAttributes();
2989}
2990
2991bool UnwrappedLineParser::handleCppAttributes() {
2992 // Handle [[likely]] / [[unlikely]] attributes.
2993 assert(FormatTok->is(tok::l_square));
2994 if (!tryToParseSimpleAttribute())
2995 return false;
2996 parseSquare();
2997 return true;
2998}
2999
3000/// Returns whether \c Tok begins a block.
3001bool UnwrappedLineParser::isBlockBegin(const FormatToken &Tok) const {
3002 // FIXME: rename the function or make
3003 // Tok.isOneOf(tok::l_brace, TT_MacroBlockBegin) work.
3004 return Style.isVerilog() ? Keywords.isVerilogBegin(Tok)
3005 : Tok.is(Kind: tok::l_brace);
3006}
3007
3008FormatToken *UnwrappedLineParser::parseIfThenElse(IfStmtKind *IfKind,
3009 bool KeepBraces,
3010 bool IsVerilogAssert) {
3011 assert((FormatTok->is(tok::kw_if) ||
3012 (Style.isVerilog() &&
3013 FormatTok->isOneOf(tok::kw_restrict, Keywords.kw_assert,
3014 Keywords.kw_assume, Keywords.kw_cover))) &&
3015 "'if' expected");
3016 nextToken();
3017
3018 if (IsVerilogAssert) {
3019 // Handle `assert #0` and `assert final`.
3020 if (FormatTok->is(II: Keywords.kw_verilogHash)) {
3021 nextToken();
3022 if (FormatTok->is(Kind: tok::numeric_constant))
3023 nextToken();
3024 } else if (FormatTok->isOneOf(K1: Keywords.kw_final, K2: Keywords.kw_property,
3025 Ks: Keywords.kw_sequence)) {
3026 nextToken();
3027 }
3028 }
3029
3030 // TableGen's if statement has the form of `if <cond> then { ... }`.
3031 if (Style.isTableGen()) {
3032 while (!eof() && FormatTok->isNot(Kind: Keywords.kw_then)) {
3033 // Simply skip until then. This range only contains a value.
3034 nextToken();
3035 }
3036 }
3037
3038 // Handle `if !consteval`.
3039 if (FormatTok->is(Kind: tok::exclaim))
3040 nextToken();
3041
3042 bool KeepIfBraces = true;
3043 if (FormatTok->is(Kind: tok::kw_consteval)) {
3044 nextToken();
3045 } else {
3046 KeepIfBraces = !Style.RemoveBracesLLVM || KeepBraces;
3047 if (FormatTok->isOneOf(K1: tok::kw_constexpr, K2: tok::identifier))
3048 nextToken();
3049 if (FormatTok->is(Kind: tok::l_paren)) {
3050 FormatTok->setFinalizedType(TT_ConditionLParen);
3051 parseParens();
3052 }
3053 }
3054 handleAttributes();
3055 // The then action is optional in Verilog assert statements.
3056 if (IsVerilogAssert && FormatTok->is(Kind: tok::semi)) {
3057 nextToken();
3058 addUnwrappedLine();
3059 return nullptr;
3060 }
3061
3062 bool NeedsUnwrappedLine = false;
3063 keepAncestorBraces();
3064
3065 FormatToken *IfLeftBrace = nullptr;
3066 IfStmtKind IfBlockKind = IfStmtKind::NotIf;
3067
3068 if (isBlockBegin(Tok: *FormatTok)) {
3069 FormatTok->setFinalizedType(TT_ControlStatementLBrace);
3070 IfLeftBrace = FormatTok;
3071 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3072 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
3073 /*MunchSemi=*/true, KeepBraces: KeepIfBraces, IfKind: &IfBlockKind);
3074 setPreviousRBraceType(TT_ControlStatementRBrace);
3075 if (Style.BraceWrapping.BeforeElse)
3076 addUnwrappedLine();
3077 else
3078 NeedsUnwrappedLine = true;
3079 } else if (IsVerilogAssert && FormatTok->is(Kind: tok::kw_else)) {
3080 addUnwrappedLine();
3081 } else {
3082 parseUnbracedBody();
3083 }
3084
3085 if (Style.RemoveBracesLLVM) {
3086 assert(!NestedTooDeep.empty());
3087 KeepIfBraces = KeepIfBraces ||
3088 (IfLeftBrace && !IfLeftBrace->MatchingParen) ||
3089 NestedTooDeep.back() || IfBlockKind == IfStmtKind::IfOnly ||
3090 IfBlockKind == IfStmtKind::IfElseIf;
3091 }
3092
3093 bool KeepElseBraces = KeepIfBraces;
3094 FormatToken *ElseLeftBrace = nullptr;
3095 IfStmtKind Kind = IfStmtKind::IfOnly;
3096
3097 if (FormatTok->is(Kind: tok::kw_else)) {
3098 if (Style.RemoveBracesLLVM) {
3099 NestedTooDeep.back() = false;
3100 Kind = IfStmtKind::IfElse;
3101 }
3102 nextToken();
3103 handleAttributes();
3104 if (isBlockBegin(Tok: *FormatTok)) {
3105 const bool FollowedByIf = Tokens->peekNextToken()->is(Kind: tok::kw_if);
3106 FormatTok->setFinalizedType(TT_ElseLBrace);
3107 ElseLeftBrace = FormatTok;
3108 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3109 IfStmtKind ElseBlockKind = IfStmtKind::NotIf;
3110 FormatToken *IfLBrace =
3111 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
3112 /*MunchSemi=*/true, KeepBraces: KeepElseBraces, IfKind: &ElseBlockKind);
3113 setPreviousRBraceType(TT_ElseRBrace);
3114 if (FormatTok->is(Kind: tok::kw_else)) {
3115 KeepElseBraces = KeepElseBraces ||
3116 ElseBlockKind == IfStmtKind::IfOnly ||
3117 ElseBlockKind == IfStmtKind::IfElseIf;
3118 } else if (FollowedByIf && IfLBrace && !IfLBrace->Optional) {
3119 KeepElseBraces = true;
3120 assert(ElseLeftBrace->MatchingParen);
3121 markOptionalBraces(LeftBrace: ElseLeftBrace);
3122 }
3123 addUnwrappedLine();
3124 } else if (!IsVerilogAssert && FormatTok->is(Kind: tok::kw_if)) {
3125 const FormatToken *Previous = Tokens->getPreviousToken();
3126 assert(Previous);
3127 const bool IsPrecededByComment = Previous->is(Kind: tok::comment);
3128 if (IsPrecededByComment) {
3129 addUnwrappedLine();
3130 ++Line->Level;
3131 }
3132 bool TooDeep = true;
3133 if (Style.RemoveBracesLLVM) {
3134 Kind = IfStmtKind::IfElseIf;
3135 TooDeep = NestedTooDeep.pop_back_val();
3136 }
3137 ElseLeftBrace = parseIfThenElse(/*IfKind=*/nullptr, KeepBraces: KeepIfBraces);
3138 if (Style.RemoveBracesLLVM)
3139 NestedTooDeep.push_back(Elt: TooDeep);
3140 if (IsPrecededByComment)
3141 --Line->Level;
3142 } else {
3143 parseUnbracedBody(/*CheckEOF=*/true);
3144 }
3145 } else {
3146 KeepIfBraces = KeepIfBraces || IfBlockKind == IfStmtKind::IfElse;
3147 if (NeedsUnwrappedLine)
3148 addUnwrappedLine();
3149 }
3150
3151 if (!Style.RemoveBracesLLVM)
3152 return nullptr;
3153
3154 assert(!NestedTooDeep.empty());
3155 KeepElseBraces = KeepElseBraces ||
3156 (ElseLeftBrace && !ElseLeftBrace->MatchingParen) ||
3157 NestedTooDeep.back();
3158
3159 NestedTooDeep.pop_back();
3160
3161 if (!KeepIfBraces && !KeepElseBraces) {
3162 markOptionalBraces(LeftBrace: IfLeftBrace);
3163 markOptionalBraces(LeftBrace: ElseLeftBrace);
3164 } else if (IfLeftBrace) {
3165 FormatToken *IfRightBrace = IfLeftBrace->MatchingParen;
3166 if (IfRightBrace) {
3167 assert(IfRightBrace->MatchingParen == IfLeftBrace);
3168 assert(!IfLeftBrace->Optional);
3169 assert(!IfRightBrace->Optional);
3170 IfLeftBrace->MatchingParen = nullptr;
3171 IfRightBrace->MatchingParen = nullptr;
3172 }
3173 }
3174
3175 if (IfKind)
3176 *IfKind = Kind;
3177
3178 return IfLeftBrace;
3179}
3180
3181void UnwrappedLineParser::parseTryCatch() {
3182 assert(FormatTok->isOneOf(tok::kw_try, tok::kw___try) && "'try' expected");
3183 nextToken();
3184 bool NeedsUnwrappedLine = false;
3185 bool HasCtorInitializer = false;
3186 if (FormatTok->is(Kind: tok::colon)) {
3187 auto *Colon = FormatTok;
3188 // We are in a function try block, what comes is an initializer list.
3189 nextToken();
3190 if (FormatTok->is(Kind: tok::identifier)) {
3191 HasCtorInitializer = true;
3192 Colon->setFinalizedType(TT_CtorInitializerColon);
3193 }
3194
3195 // In case identifiers were removed by clang-tidy, what might follow is
3196 // multiple commas in sequence - before the first identifier.
3197 while (FormatTok->is(Kind: tok::comma))
3198 nextToken();
3199
3200 while (FormatTok->is(Kind: tok::identifier)) {
3201 nextToken();
3202 if (FormatTok->is(Kind: tok::l_paren)) {
3203 parseParens();
3204 } else if (FormatTok->is(Kind: tok::l_brace)) {
3205 nextToken();
3206 parseBracedList();
3207 }
3208
3209 // In case identifiers were removed by clang-tidy, what might follow is
3210 // multiple commas in sequence - after the first identifier.
3211 while (FormatTok->is(Kind: tok::comma))
3212 nextToken();
3213 }
3214 }
3215 // Parse try with resource.
3216 if (Style.isJava() && FormatTok->is(Kind: tok::l_paren))
3217 parseParens();
3218
3219 keepAncestorBraces();
3220
3221 if (FormatTok->is(Kind: tok::l_brace)) {
3222 if (HasCtorInitializer)
3223 FormatTok->setFinalizedType(TT_FunctionLBrace);
3224 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3225 parseBlock();
3226 if (Style.BraceWrapping.BeforeCatch)
3227 addUnwrappedLine();
3228 else
3229 NeedsUnwrappedLine = true;
3230 } else if (FormatTok->isNot(Kind: tok::kw_catch)) {
3231 // The C++ standard requires a compound-statement after a try.
3232 // If there's none, we try to assume there's a structuralElement
3233 // and try to continue.
3234 addUnwrappedLine();
3235 ++Line->Level;
3236 parseStructuralElement();
3237 --Line->Level;
3238 }
3239 for (bool SeenCatch = false;;) {
3240 if (FormatTok->is(Kind: tok::at))
3241 nextToken();
3242 if (FormatTok->isNoneOf(Ks: tok::kw_catch, Ks: Keywords.kw___except,
3243 Ks: tok::kw___finally, Ks: tok::objc_catch,
3244 Ks: tok::objc_finally) &&
3245 !((Style.isJava() || Style.isJavaScript()) &&
3246 FormatTok->is(II: Keywords.kw_finally))) {
3247 break;
3248 }
3249 if (FormatTok->is(Kind: tok::kw_catch))
3250 SeenCatch = true;
3251 nextToken();
3252 while (FormatTok->isNot(Kind: tok::l_brace)) {
3253 if (FormatTok->is(Kind: tok::l_paren)) {
3254 parseParens();
3255 continue;
3256 }
3257 if (FormatTok->isOneOf(K1: tok::semi, K2: tok::r_brace) || eof()) {
3258 if (Style.RemoveBracesLLVM)
3259 NestedTooDeep.pop_back();
3260 return;
3261 }
3262 nextToken();
3263 }
3264 if (SeenCatch) {
3265 FormatTok->setFinalizedType(TT_ControlStatementLBrace);
3266 SeenCatch = false;
3267 }
3268 NeedsUnwrappedLine = false;
3269 Line->MustBeDeclaration = false;
3270 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3271 parseBlock();
3272 if (Style.BraceWrapping.BeforeCatch)
3273 addUnwrappedLine();
3274 else
3275 NeedsUnwrappedLine = true;
3276 }
3277
3278 if (Style.RemoveBracesLLVM)
3279 NestedTooDeep.pop_back();
3280
3281 if (NeedsUnwrappedLine)
3282 addUnwrappedLine();
3283}
3284
3285void UnwrappedLineParser::parseNamespaceOrExportBlock(unsigned AddLevels) {
3286 bool ManageWhitesmithsBraces =
3287 AddLevels == 0u && Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths;
3288
3289 // If we're in Whitesmiths mode, indent the brace if we're not indenting
3290 // the whole block.
3291 if (ManageWhitesmithsBraces)
3292 ++Line->Level;
3293
3294 // Munch the semicolon after the block. This is more common than one would
3295 // think. Putting the semicolon into its own line is very ugly.
3296 parseBlock(/*MustBeDeclaration=*/true, AddLevels, /*MunchSemi=*/true,
3297 /*KeepBraces=*/true, /*IfKind=*/nullptr, UnindentWhitesmithsBraces: ManageWhitesmithsBraces);
3298
3299 addUnwrappedLine(AdjustLevel: AddLevels > 0 ? LineLevel::Remove : LineLevel::Keep);
3300
3301 if (ManageWhitesmithsBraces)
3302 --Line->Level;
3303}
3304
3305void UnwrappedLineParser::parseNamespace() {
3306 assert(FormatTok->isOneOf(tok::kw_namespace, TT_NamespaceMacro) &&
3307 "'namespace' expected");
3308
3309 const FormatToken &InitialToken = *FormatTok;
3310 nextToken();
3311 if (InitialToken.is(TT: TT_NamespaceMacro)) {
3312 parseParens();
3313 } else {
3314 while (FormatTok->isOneOf(K1: tok::identifier, K2: tok::coloncolon, Ks: tok::kw_inline,
3315 Ks: tok::l_square, Ks: tok::period, Ks: tok::l_paren) ||
3316 (Style.isCSharp() && FormatTok->is(Kind: tok::kw_union))) {
3317 if (FormatTok->is(Kind: tok::l_square))
3318 parseSquare();
3319 else if (FormatTok->is(Kind: tok::l_paren))
3320 parseParens();
3321 else
3322 nextToken();
3323 }
3324 }
3325 if (FormatTok->is(Kind: tok::l_brace)) {
3326 FormatTok->setFinalizedType(TT_NamespaceLBrace);
3327
3328 if (ShouldBreakBeforeBrace(Style, InitialToken,
3329 IsEmptyBlock: Tokens->peekNextToken()->is(Kind: tok::r_brace))) {
3330 addUnwrappedLine();
3331 }
3332
3333 unsigned AddLevels =
3334 Style.NamespaceIndentation == FormatStyle::NI_All ||
3335 (Style.NamespaceIndentation == FormatStyle::NI_Inner &&
3336 DeclarationScopeStack.size() > 1)
3337 ? 1u
3338 : 0u;
3339 parseNamespaceOrExportBlock(AddLevels);
3340 }
3341 // FIXME: Add error handling.
3342}
3343
3344void UnwrappedLineParser::parseCppExportBlock() {
3345 if (FormatTok->is(Kind: tok::l_brace)) {
3346 FormatTok->setFinalizedType(TT_ExportLBrace);
3347 if (Style.BraceWrapping.AfterExportBlock)
3348 addUnwrappedLine();
3349 }
3350 parseNamespaceOrExportBlock(/*AddLevels=*/Style.IndentExportBlock ? 1 : 0);
3351}
3352
3353void UnwrappedLineParser::parseNew() {
3354 assert(FormatTok->is(tok::kw_new) && "'new' expected");
3355 nextToken();
3356
3357 if (Style.isCSharp()) {
3358 do {
3359 // Handle constructor invocation, e.g. `new(field: value)`.
3360 if (FormatTok->is(Kind: tok::l_paren))
3361 parseParens();
3362
3363 // Handle array initialization syntax, e.g. `new[] {10, 20, 30}`.
3364 if (FormatTok->is(Kind: tok::l_brace))
3365 parseBracedList();
3366
3367 if (FormatTok->isOneOf(K1: tok::semi, K2: tok::comma))
3368 return;
3369
3370 nextToken();
3371 } while (!eof());
3372 }
3373
3374 if (!Style.isJava())
3375 return;
3376
3377 // In Java, we can parse everything up to the parens, which aren't optional.
3378 do {
3379 // There should not be a ;, { or } before the new's open paren.
3380 if (FormatTok->isOneOf(K1: tok::semi, K2: tok::l_brace, Ks: tok::r_brace))
3381 return;
3382
3383 // Consume the parens.
3384 if (FormatTok->is(Kind: tok::l_paren)) {
3385 parseParens();
3386
3387 // If there is a class body of an anonymous class, consume that as child.
3388 if (FormatTok->is(Kind: tok::l_brace))
3389 parseChildBlock();
3390 return;
3391 }
3392 nextToken();
3393 } while (!eof());
3394}
3395
3396void UnwrappedLineParser::parseLoopBody(bool KeepBraces, bool WrapRightBrace) {
3397 keepAncestorBraces();
3398
3399 if (isBlockBegin(Tok: *FormatTok)) {
3400 FormatTok->setFinalizedType(TT_ControlStatementLBrace);
3401 FormatToken *LeftBrace = FormatTok;
3402 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3403 parseBlock(/*MustBeDeclaration=*/false, /*AddLevels=*/1u,
3404 /*MunchSemi=*/true, KeepBraces);
3405 setPreviousRBraceType(TT_ControlStatementRBrace);
3406 if (!KeepBraces) {
3407 assert(!NestedTooDeep.empty());
3408 if (!NestedTooDeep.back())
3409 markOptionalBraces(LeftBrace);
3410 }
3411 if (WrapRightBrace)
3412 addUnwrappedLine();
3413 } else {
3414 parseUnbracedBody();
3415 }
3416
3417 if (!KeepBraces)
3418 NestedTooDeep.pop_back();
3419}
3420
3421void UnwrappedLineParser::parseForOrWhileLoop(bool HasParens) {
3422 assert((FormatTok->isOneOf(tok::kw_for, tok::kw_while, TT_ForEachMacro) ||
3423 (Style.isVerilog() &&
3424 FormatTok->isOneOf(Keywords.kw_always, Keywords.kw_always_comb,
3425 Keywords.kw_always_ff, Keywords.kw_always_latch,
3426 Keywords.kw_final, Keywords.kw_initial,
3427 Keywords.kw_foreach, Keywords.kw_forever,
3428 Keywords.kw_repeat))) &&
3429 "'for', 'while' or foreach macro expected");
3430 const bool KeepBraces = !Style.RemoveBracesLLVM ||
3431 FormatTok->isNoneOf(Ks: tok::kw_for, Ks: tok::kw_while);
3432
3433 nextToken();
3434 // JS' for await ( ...
3435 if (Style.isJavaScript() && FormatTok->is(II: Keywords.kw_await))
3436 nextToken();
3437 if (IsCpp && FormatTok->is(Kind: tok::kw_co_await))
3438 nextToken();
3439 if (HasParens && FormatTok->is(Kind: tok::l_paren)) {
3440 // The type is only set for Verilog basically because we were afraid to
3441 // change the existing behavior for loops. See the discussion on D121756 for
3442 // details.
3443 if (Style.isVerilog())
3444 FormatTok->setFinalizedType(TT_ConditionLParen);
3445 parseParens();
3446 }
3447
3448 if (Style.isVerilog()) {
3449 // Event control.
3450 parseVerilogSensitivityList();
3451 } else if (Style.AllowShortLoopsOnASingleLine && FormatTok->is(Kind: tok::semi) &&
3452 Tokens->getPreviousToken()->is(Kind: tok::r_paren)) {
3453 nextToken();
3454 addUnwrappedLine();
3455 return;
3456 }
3457
3458 handleAttributes();
3459 parseLoopBody(KeepBraces, /*WrapRightBrace=*/true);
3460}
3461
3462void UnwrappedLineParser::parseDoWhile() {
3463 assert(FormatTok->is(tok::kw_do) && "'do' expected");
3464 nextToken();
3465
3466 parseLoopBody(/*KeepBraces=*/true, WrapRightBrace: Style.BraceWrapping.BeforeWhile);
3467
3468 // FIXME: Add error handling.
3469 if (FormatTok->isNot(Kind: tok::kw_while)) {
3470 addUnwrappedLine();
3471 return;
3472 }
3473
3474 FormatTok->setFinalizedType(TT_DoWhile);
3475
3476 // If in Whitesmiths mode, the line with the while() needs to be indented
3477 // to the same level as the block.
3478 if (Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths)
3479 ++Line->Level;
3480
3481 nextToken();
3482 parseStructuralElement();
3483}
3484
3485void UnwrappedLineParser::parseLabel(bool IsGotoLabel) {
3486 nextToken();
3487
3488 const auto IndentGotoLabel = Style.IndentGotoLabels;
3489 const auto OldLineLevel = Line->Level;
3490 auto &Level = Line->Level;
3491
3492 if (IsGotoLabel && IndentGotoLabel == FormatStyle::IGLS_NoIndent)
3493 Level = 0;
3494
3495 if (!IsGotoLabel || IndentGotoLabel == FormatStyle::IGLS_OuterIndent) {
3496 if (OldLineLevel > 1 || (!Line->InPPDirective && OldLineLevel > 0))
3497 --Level;
3498 }
3499
3500 if (!IsGotoLabel && !Style.IndentCaseBlocks &&
3501 CommentsBeforeNextToken.empty() && FormatTok->is(Kind: tok::l_brace)) {
3502 CompoundStatementIndenter Indenter(this, Level,
3503 Style.BraceWrapping.AfterCaseLabel,
3504 Style.BraceWrapping.IndentBraces);
3505 parseBlock();
3506 if (FormatTok->is(Kind: tok::kw_break)) {
3507 if (Style.BraceWrapping.AfterControlStatement ==
3508 FormatStyle::BWACS_Always) {
3509 addUnwrappedLine();
3510 if (!Style.IndentCaseBlocks &&
3511 Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths) {
3512 ++Level;
3513 }
3514 }
3515 parseStructuralElement();
3516 }
3517 addUnwrappedLine();
3518 } else {
3519 if (FormatTok->is(Kind: tok::semi))
3520 nextToken();
3521 addUnwrappedLine();
3522 }
3523
3524 Level = OldLineLevel;
3525
3526 if (FormatTok->isNot(Kind: tok::l_brace)) {
3527 parseStructuralElement();
3528 addUnwrappedLine();
3529 }
3530}
3531
3532void UnwrappedLineParser::parseCaseLabel() {
3533 assert(FormatTok->is(tok::kw_case) && "'case' expected");
3534 auto *Case = FormatTok;
3535
3536 // FIXME: fix handling of complex expressions here.
3537 do {
3538 nextToken();
3539 if (FormatTok->is(Kind: tok::colon)) {
3540 FormatTok->setFinalizedType(TT_CaseLabelColon);
3541 break;
3542 }
3543 if (Style.isJava() && FormatTok->is(Kind: tok::arrow)) {
3544 FormatTok->setFinalizedType(TT_CaseLabelArrow);
3545 Case->setFinalizedType(TT_SwitchExpressionLabel);
3546 break;
3547 }
3548 } while (!eof());
3549 parseLabel();
3550}
3551
3552void UnwrappedLineParser::parseSwitch(bool IsExpr) {
3553 assert(FormatTok->is(tok::kw_switch) && "'switch' expected");
3554 nextToken();
3555 if (FormatTok->is(Kind: tok::l_paren))
3556 parseParens();
3557
3558 keepAncestorBraces();
3559
3560 if (FormatTok->is(Kind: tok::l_brace)) {
3561 CompoundStatementIndenter Indenter(this, Style, Line->Level);
3562 FormatTok->setFinalizedType(IsExpr ? TT_SwitchExpressionLBrace
3563 : TT_ControlStatementLBrace);
3564 if (IsExpr)
3565 parseChildBlock();
3566 else
3567 parseBlock();
3568 setPreviousRBraceType(TT_ControlStatementRBrace);
3569 if (!IsExpr)
3570 addUnwrappedLine();
3571 } else {
3572 addUnwrappedLine();
3573 ++Line->Level;
3574 parseStructuralElement();
3575 --Line->Level;
3576 }
3577
3578 if (Style.RemoveBracesLLVM)
3579 NestedTooDeep.pop_back();
3580}
3581
3582void UnwrappedLineParser::parseAccessSpecifier() {
3583 nextToken();
3584 // Understand Qt's slots.
3585 if (FormatTok->isOneOf(K1: Keywords.kw_slots, K2: Keywords.kw_qslots))
3586 nextToken();
3587 // Otherwise, we don't know what it is, and we'd better keep the next token.
3588 if (FormatTok->is(Kind: tok::colon))
3589 nextToken();
3590 addUnwrappedLine();
3591}
3592
3593/// Parses a requires, decides if it is a clause or an expression.
3594/// \pre The current token has to be the requires keyword.
3595/// \returns true if it parsed a clause.
3596bool UnwrappedLineParser::parseRequires(bool SeenEqual) {
3597 assert(FormatTok->is(tok::kw_requires) && "'requires' expected");
3598
3599 // We try to guess if it is a requires clause, or a requires expression. For
3600 // that we first check the next token.
3601 switch (Tokens->peekNextToken(/*SkipComment=*/true)->Tok.getKind()) {
3602 case tok::l_brace:
3603 // This can only be an expression, never a clause.
3604 parseRequiresExpression();
3605 return false;
3606 case tok::l_paren:
3607 // Clauses and expression can start with a paren, it's unclear what we have.
3608 break;
3609 default:
3610 // All other tokens can only be a clause.
3611 parseRequiresClause();
3612 return true;
3613 }
3614
3615 // Looking forward we would have to decide if there are function declaration
3616 // like arguments to the requires expression:
3617 // requires (T t) {
3618 // Or there is a constraint expression for the requires clause:
3619 // requires (C<T> && ...
3620
3621 // But first let's look behind.
3622 auto *PreviousNonComment = FormatTok->getPreviousNonComment();
3623
3624 if (!PreviousNonComment ||
3625 PreviousNonComment->is(TT: TT_RequiresExpressionLBrace)) {
3626 // If there is no token, or an expression left brace, we are a requires
3627 // clause within a requires expression.
3628 parseRequiresClause();
3629 return true;
3630 }
3631
3632 switch (PreviousNonComment->Tok.getKind()) {
3633 case tok::greater:
3634 case tok::r_paren:
3635 case tok::kw_noexcept:
3636 case tok::kw_const:
3637 case tok::star:
3638 case tok::amp:
3639 // This is a requires clause.
3640 parseRequiresClause();
3641 return true;
3642 case tok::ampamp: {
3643 // This can be either:
3644 // if (... && requires (T t) ...)
3645 // Or
3646 // void member(...) && requires (C<T> ...
3647 // We check the one token before that for a const:
3648 // void member(...) const && requires (C<T> ...
3649 auto PrevPrev = PreviousNonComment->getPreviousNonComment();
3650 if ((PrevPrev && PrevPrev->is(Kind: tok::kw_const)) || !SeenEqual) {
3651 parseRequiresClause();
3652 return true;
3653 }
3654 break;
3655 }
3656 default:
3657 if (PreviousNonComment->isTypeOrIdentifier(LangOpts)) {
3658 // This is a requires clause.
3659 parseRequiresClause();
3660 return true;
3661 }
3662 // It's an expression.
3663 parseRequiresExpression();
3664 return false;
3665 }
3666
3667 // Now we look forward and try to check if the paren content is a parameter
3668 // list. The parameters can be cv-qualified and contain references or
3669 // pointers.
3670 // So we want basically to check for TYPE NAME, but TYPE can contain all kinds
3671 // of stuff: typename, const, *, &, &&, ::, identifiers.
3672
3673 unsigned StoredPosition = Tokens->getPosition();
3674 FormatToken *NextToken = Tokens->getNextToken();
3675 int Lookahead = 0;
3676 auto PeekNext = [&Lookahead, &NextToken, this] {
3677 ++Lookahead;
3678 NextToken = Tokens->getNextToken();
3679 };
3680
3681 bool FoundType = false;
3682 bool LastWasColonColon = false;
3683 int OpenAngles = 0;
3684
3685 for (; Lookahead < 50; PeekNext()) {
3686 switch (NextToken->Tok.getKind()) {
3687 case tok::kw_volatile:
3688 case tok::kw_const:
3689 case tok::comma:
3690 if (OpenAngles == 0) {
3691 FormatTok = Tokens->setPosition(StoredPosition);
3692 parseRequiresExpression();
3693 return false;
3694 }
3695 break;
3696 case tok::eof:
3697 // Break out of the loop.
3698 Lookahead = 50;
3699 break;
3700 case tok::coloncolon:
3701 LastWasColonColon = true;
3702 break;
3703 case tok::kw_decltype:
3704 case tok::identifier:
3705 if (FoundType && !LastWasColonColon && OpenAngles == 0) {
3706 FormatTok = Tokens->setPosition(StoredPosition);
3707 parseRequiresExpression();
3708 return false;
3709 }
3710 FoundType = true;
3711 LastWasColonColon = false;
3712 break;
3713 case tok::less:
3714 ++OpenAngles;
3715 break;
3716 case tok::greater:
3717 --OpenAngles;
3718 break;
3719 default:
3720 if (NextToken->isTypeName(LangOpts)) {
3721 FormatTok = Tokens->setPosition(StoredPosition);
3722 parseRequiresExpression();
3723 return false;
3724 }
3725 break;
3726 }
3727 }
3728 // This seems to be a complicated expression, just assume it's a clause.
3729 FormatTok = Tokens->setPosition(StoredPosition);
3730 parseRequiresClause();
3731 return true;
3732}
3733
3734/// Parses a requires clause.
3735/// \sa parseRequiresExpression
3736///
3737/// Returns if it either has finished parsing the clause, or it detects, that
3738/// the clause is incorrect.
3739void UnwrappedLineParser::parseRequiresClause() {
3740 assert(FormatTok->is(tok::kw_requires) && "'requires' expected");
3741
3742 // If there is no previous token, we are within a requires expression,
3743 // otherwise we will always have the template or function declaration in front
3744 // of it.
3745 bool InRequiresExpression =
3746 !FormatTok->Previous ||
3747 FormatTok->Previous->is(TT: TT_RequiresExpressionLBrace);
3748
3749 FormatTok->setFinalizedType(InRequiresExpression
3750 ? TT_RequiresClauseInARequiresExpression
3751 : TT_RequiresClause);
3752 nextToken();
3753
3754 // NOTE: parseConstraintExpression is only ever called from this function.
3755 // It could be inlined into here.
3756 parseConstraintExpression();
3757
3758 if (!InRequiresExpression && FormatTok->Previous)
3759 FormatTok->Previous->ClosesRequiresClause = true;
3760}
3761
3762/// Parses a requires expression.
3763/// \sa parseRequiresClause
3764///
3765/// Returns if it either has finished parsing the expression, or it detects,
3766/// that the expression is incorrect.
3767void UnwrappedLineParser::parseRequiresExpression() {
3768 assert(FormatTok->is(tok::kw_requires) && "'requires' expected");
3769
3770 FormatTok->setFinalizedType(TT_RequiresExpression);
3771 nextToken();
3772
3773 if (FormatTok->is(Kind: tok::l_paren)) {
3774 FormatTok->setFinalizedType(TT_RequiresExpressionLParen);
3775 parseParens();
3776 }
3777
3778 if (FormatTok->is(Kind: tok::l_brace)) {
3779 FormatTok->setFinalizedType(TT_RequiresExpressionLBrace);
3780 parseChildBlock();
3781 }
3782}
3783
3784/// Parses a constraint expression.
3785///
3786/// This is the body of a requires clause. It returns, when the parsing is
3787/// complete, or the expression is incorrect.
3788void UnwrappedLineParser::parseConstraintExpression() {
3789 // The special handling for lambdas is needed since tryToParseLambda() eats a
3790 // token and if a requires expression is the last part of a requires clause
3791 // and followed by an attribute like [[nodiscard]] the ClosesRequiresClause is
3792 // not set on the correct token. Thus we need to be aware if we even expect a
3793 // lambda to be possible.
3794 // template <typename T> requires requires { ... } [[nodiscard]] ...;
3795 bool LambdaNextTimeAllowed = true;
3796
3797 // Within lambda declarations, it is permitted to put a requires clause after
3798 // its template parameter list, which would place the requires clause right
3799 // before the parentheses of the parameters of the lambda declaration. Thus,
3800 // we track if we expect to see grouping parentheses at all.
3801 // Without this check, `requires foo<T> (T t)` in the below example would be
3802 // seen as the whole requires clause, accidentally eating the parameters of
3803 // the lambda.
3804 // [&]<typename T> requires foo<T> (T t) { ... };
3805 bool TopLevelParensAllowed = true;
3806
3807 do {
3808 bool LambdaThisTimeAllowed = std::exchange(obj&: LambdaNextTimeAllowed, new_val: false);
3809
3810 switch (FormatTok->Tok.getKind()) {
3811 case tok::kw_requires:
3812 parseRequiresExpression();
3813 break;
3814
3815 case tok::l_paren:
3816 if (!TopLevelParensAllowed)
3817 return;
3818 parseParens(/*AmpAmpTokenType=*/StarAndAmpTokenType: TT_BinaryOperator);
3819 TopLevelParensAllowed = false;
3820 break;
3821
3822 case tok::l_square:
3823 if (!LambdaThisTimeAllowed || !tryToParseLambda())
3824 return;
3825 break;
3826
3827 case tok::kw_const:
3828 case tok::semi:
3829 case tok::kw_class:
3830 case tok::kw_struct:
3831 case tok::kw_union:
3832 return;
3833
3834 case tok::l_brace:
3835 // Potential function body.
3836 return;
3837
3838 case tok::ampamp:
3839 case tok::pipepipe:
3840 FormatTok->setFinalizedType(TT_BinaryOperator);
3841 nextToken();
3842 LambdaNextTimeAllowed = true;
3843 TopLevelParensAllowed = true;
3844 break;
3845
3846 case tok::comma:
3847 case tok::comment:
3848 LambdaNextTimeAllowed = LambdaThisTimeAllowed;
3849 nextToken();
3850 break;
3851
3852 case tok::kw_sizeof:
3853 case tok::greater:
3854 case tok::greaterequal:
3855 case tok::greatergreater:
3856 case tok::less:
3857 case tok::lessequal:
3858 case tok::lessless:
3859 case tok::equalequal:
3860 case tok::exclaim:
3861 case tok::exclaimequal:
3862 case tok::plus:
3863 case tok::minus:
3864 case tok::star:
3865 case tok::slash:
3866 LambdaNextTimeAllowed = true;
3867 TopLevelParensAllowed = true;
3868 // Just eat them.
3869 nextToken();
3870 break;
3871
3872 case tok::numeric_constant:
3873 case tok::coloncolon:
3874 case tok::kw_true:
3875 case tok::kw_false:
3876 TopLevelParensAllowed = false;
3877 // Just eat them.
3878 nextToken();
3879 break;
3880
3881 case tok::kw_static_cast:
3882 case tok::kw_const_cast:
3883 case tok::kw_reinterpret_cast:
3884 case tok::kw_dynamic_cast:
3885 nextToken();
3886 if (FormatTok->isNot(Kind: tok::less))
3887 return;
3888
3889 nextToken();
3890 parseBracedList(/*IsAngleBracket=*/true);
3891 break;
3892
3893 default:
3894 if (!FormatTok->Tok.getIdentifierInfo()) {
3895 // Identifiers are part of the default case, we check for more then
3896 // tok::identifier to handle builtin type traits.
3897 return;
3898 }
3899
3900 // We need to differentiate identifiers for a template deduction guide,
3901 // variables, or function return types (the constraint expression has
3902 // ended before that), and basically all other cases. But it's easier to
3903 // check the other way around.
3904 assert(FormatTok->Previous);
3905 switch (FormatTok->Previous->Tok.getKind()) {
3906 case tok::coloncolon: // Nested identifier.
3907 case tok::ampamp: // Start of a function or variable for the
3908 case tok::pipepipe: // constraint expression. (binary)
3909 case tok::exclaim: // The same as above, but unary.
3910 case tok::kw_requires: // Initial identifier of a requires clause.
3911 case tok::equal: // Initial identifier of a concept declaration.
3912 case tok::kw_template: // A dependent template.
3913 break;
3914 default:
3915 return;
3916 }
3917
3918 // Read identifier with optional template declaration.
3919 nextToken();
3920 if (FormatTok->is(Kind: tok::less)) {
3921 nextToken();
3922 parseBracedList(/*IsAngleBracket=*/true);
3923 }
3924 TopLevelParensAllowed = false;
3925 break;
3926 }
3927 } while (!eof());
3928}
3929
3930bool UnwrappedLineParser::parseEnum() {
3931 const FormatToken &InitialToken = *FormatTok;
3932
3933 // Won't be 'enum' for NS_ENUMs.
3934 if (FormatTok->is(Kind: tok::kw_enum))
3935 nextToken();
3936
3937 // In TypeScript, "enum" can also be used as property name, e.g. in interface
3938 // declarations. An "enum" keyword followed by a colon would be a syntax
3939 // error and thus assume it is just an identifier.
3940 if (Style.isJavaScript() && FormatTok->isOneOf(K1: tok::colon, K2: tok::question))
3941 return false;
3942
3943 // In protobuf, "enum" can be used as a field name.
3944 if (Style.Language == FormatStyle::LK_Proto && FormatTok->is(Kind: tok::equal))
3945 return false;
3946
3947 if (IsCpp) {
3948 // Eat up enum class ...
3949 if (FormatTok->isOneOf(K1: tok::kw_class, K2: tok::kw_struct))
3950 nextToken();
3951 while (FormatTok->is(Kind: tok::l_square))
3952 if (!handleCppAttributes())
3953 return false;
3954 }
3955
3956 while (FormatTok->Tok.getIdentifierInfo() ||
3957 FormatTok->isOneOf(K1: tok::colon, K2: tok::coloncolon, Ks: tok::less,
3958 Ks: tok::greater, Ks: tok::comma, Ks: tok::question,
3959 Ks: tok::l_square)) {
3960 if (FormatTok->is(Kind: tok::colon))
3961 FormatTok->setFinalizedType(TT_EnumUnderlyingTypeColon);
3962 if (Style.isVerilog()) {
3963 FormatTok->setFinalizedType(TT_VerilogDimensionedTypeName);
3964 nextToken();
3965 // In Verilog the base type can have dimensions.
3966 while (FormatTok->is(Kind: tok::l_square))
3967 parseSquare();
3968 } else {
3969 nextToken();
3970 }
3971 // We can have macros or attributes in between 'enum' and the enum name.
3972 if (FormatTok->is(Kind: tok::l_paren))
3973 parseParens();
3974 if (FormatTok->is(Kind: tok::identifier)) {
3975 nextToken();
3976 // If there are two identifiers in a row, this is likely an elaborate
3977 // return type. In Java, this can be "implements", etc.
3978 if (IsCpp && FormatTok->is(Kind: tok::identifier))
3979 return false;
3980 }
3981 }
3982
3983 // Just a declaration or something is wrong.
3984 if (FormatTok->isNot(Kind: tok::l_brace))
3985 return true;
3986 FormatTok->setFinalizedType(TT_EnumLBrace);
3987 FormatTok->setBlockKind(BK_Block);
3988
3989 if (Style.isJava()) {
3990 // Java enums are different.
3991 parseJavaEnumBody();
3992 return true;
3993 }
3994 if (Style.Language == FormatStyle::LK_Proto) {
3995 parseBlock(/*MustBeDeclaration=*/true);
3996 return true;
3997 }
3998
3999 const bool ManageWhitesmithsBraces =
4000 Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths;
4001
4002 if (!Style.AllowShortEnumsOnASingleLine &&
4003 ShouldBreakBeforeBrace(Style, InitialToken,
4004 IsEmptyBlock: Tokens->peekNextToken()->is(Kind: tok::r_brace))) {
4005 addUnwrappedLine();
4006
4007 // If we're in Whitesmiths mode, indent the brace if we're not indenting
4008 // the whole block.
4009 if (ManageWhitesmithsBraces)
4010 ++Line->Level;
4011 }
4012 // Parse enum body.
4013 nextToken();
4014 if (!Style.AllowShortEnumsOnASingleLine) {
4015 addUnwrappedLine();
4016 if (!ManageWhitesmithsBraces)
4017 ++Line->Level;
4018 }
4019 const auto OpeningLineIndex = CurrentLines->empty()
4020 ? UnwrappedLine::kInvalidIndex
4021 : CurrentLines->size() - 1;
4022 bool HasError = !parseBracedList(/*IsAngleBracket=*/false, /*IsEnum=*/true);
4023 if (!Style.AllowShortEnumsOnASingleLine && !ManageWhitesmithsBraces)
4024 --Line->Level;
4025 if (HasError) {
4026 if (FormatTok->is(Kind: tok::semi))
4027 nextToken();
4028 addUnwrappedLine();
4029 }
4030 setPreviousRBraceType(TT_EnumRBrace);
4031 if (ManageWhitesmithsBraces)
4032 Line->MatchingOpeningBlockLineIndex = OpeningLineIndex;
4033 return true;
4034
4035 // There is no addUnwrappedLine() here so that we fall through to parsing a
4036 // structural element afterwards. Thus, in "enum A {} n, m;",
4037 // "} n, m;" will end up in one unwrapped line.
4038}
4039
4040bool UnwrappedLineParser::parseStructLike() {
4041 // parseRecord falls through and does not yet add an unwrapped line as a
4042 // record declaration or definition can start a structural element.
4043 parseRecord();
4044 // This does not apply to Java, JavaScript and C#.
4045 if (Style.isJava() || Style.isJavaScript() || Style.isCSharp()) {
4046 if (FormatTok->is(Kind: tok::semi))
4047 nextToken();
4048 addUnwrappedLine();
4049 return true;
4050 }
4051 return false;
4052}
4053
4054namespace {
4055// A class used to set and restore the Token position when peeking
4056// ahead in the token source.
4057class ScopedTokenPosition {
4058 unsigned StoredPosition;
4059 FormatTokenSource *Tokens;
4060
4061public:
4062 ScopedTokenPosition(FormatTokenSource *Tokens) : Tokens(Tokens) {
4063 assert(Tokens && "Tokens expected to not be null");
4064 StoredPosition = Tokens->getPosition();
4065 }
4066
4067 ~ScopedTokenPosition() { Tokens->setPosition(StoredPosition); }
4068};
4069} // namespace
4070
4071// Look to see if we have [[ by looking ahead, if
4072// its not then rewind to the original position.
4073bool UnwrappedLineParser::tryToParseSimpleAttribute() {
4074 ScopedTokenPosition AutoPosition(Tokens);
4075 FormatToken *Tok = Tokens->getNextToken();
4076 // We already read the first [ check for the second.
4077 if (Tok->isNot(Kind: tok::l_square))
4078 return false;
4079 // Double check that the attribute is just something
4080 // fairly simple.
4081 while (Tok->isNot(Kind: tok::eof)) {
4082 if (Tok->is(Kind: tok::r_square))
4083 break;
4084 Tok = Tokens->getNextToken();
4085 }
4086 if (Tok->is(Kind: tok::eof))
4087 return false;
4088 Tok = Tokens->getNextToken();
4089 if (Tok->isNot(Kind: tok::r_square))
4090 return false;
4091 Tok = Tokens->getNextToken();
4092 if (Tok->is(Kind: tok::semi))
4093 return false;
4094 return true;
4095}
4096
4097void UnwrappedLineParser::parseJavaEnumBody() {
4098 assert(FormatTok->is(tok::l_brace));
4099 const FormatToken *OpeningBrace = FormatTok;
4100
4101 // Determine whether the enum is simple, i.e. does not have a semicolon or
4102 // constants with class bodies. Simple enums can be formatted like braced
4103 // lists, contracted to a single line, etc.
4104 unsigned StoredPosition = Tokens->getPosition();
4105 bool IsSimple = true;
4106 FormatToken *Tok = Tokens->getNextToken();
4107 while (Tok->isNot(Kind: tok::eof)) {
4108 if (Tok->is(Kind: tok::r_brace))
4109 break;
4110 if (Tok->isOneOf(K1: tok::l_brace, K2: tok::semi)) {
4111 IsSimple = false;
4112 break;
4113 }
4114 // FIXME: This will also mark enums with braces in the arguments to enum
4115 // constants as "not simple". This is probably fine in practice, though.
4116 Tok = Tokens->getNextToken();
4117 }
4118 FormatTok = Tokens->setPosition(StoredPosition);
4119
4120 if (IsSimple) {
4121 nextToken();
4122 parseBracedList();
4123 addUnwrappedLine();
4124 return;
4125 }
4126
4127 // Parse the body of a more complex enum.
4128 // First add a line for everything up to the "{".
4129 nextToken();
4130 addUnwrappedLine();
4131 ++Line->Level;
4132
4133 // Parse the enum constants.
4134 while (!eof()) {
4135 if (FormatTok->is(Kind: tok::l_brace)) {
4136 // Parse the constant's class body.
4137 parseBlock(/*MustBeDeclaration=*/true, /*AddLevels=*/1u,
4138 /*MunchSemi=*/false);
4139 } else if (FormatTok->is(Kind: tok::l_paren)) {
4140 parseParens();
4141 } else if (FormatTok->is(Kind: tok::comma)) {
4142 nextToken();
4143 addUnwrappedLine();
4144 } else if (FormatTok->is(Kind: tok::semi)) {
4145 nextToken();
4146 addUnwrappedLine();
4147 break;
4148 } else if (FormatTok->is(Kind: tok::r_brace)) {
4149 addUnwrappedLine();
4150 break;
4151 } else {
4152 nextToken();
4153 }
4154 }
4155
4156 // Parse the class body after the enum's ";" if any.
4157 parseLevel(OpeningBrace);
4158 nextToken();
4159 --Line->Level;
4160 addUnwrappedLine();
4161}
4162
4163void UnwrappedLineParser::parseRecord(bool ParseAsExpr, bool IsJavaRecord) {
4164 assert(!IsJavaRecord || FormatTok->is(Keywords.kw_record));
4165 const FormatToken &InitialToken = *FormatTok;
4166 nextToken();
4167
4168 FormatToken *ClassName =
4169 IsJavaRecord && FormatTok->is(Kind: tok::identifier) ? FormatTok : nullptr;
4170 bool IsDerived = false;
4171 auto IsNonMacroIdentifier = [](const FormatToken *Tok) {
4172 return Tok->is(Kind: tok::identifier) && Tok->TokenText != Tok->TokenText.upper();
4173 };
4174 // JavaScript/TypeScript supports anonymous classes like:
4175 // a = class extends foo { }
4176 bool JSPastExtendsOrImplements = false;
4177 // The actual identifier can be a nested name specifier, and in macros
4178 // it is often token-pasted.
4179 // An [[attribute]] can be before the identifier.
4180 while (FormatTok->isOneOf(K1: tok::identifier, K2: tok::coloncolon, Ks: tok::hashhash,
4181 Ks: tok::kw_alignas, Ks: tok::l_square) ||
4182 FormatTok->isAttribute() ||
4183 ((Style.isJava() || Style.isJavaScript()) &&
4184 FormatTok->isOneOf(K1: tok::period, K2: tok::comma)) ||
4185 (Style.isVerilog() &&
4186 FormatTok->isOneOf(K1: tok::kw_signed, K2: tok::kw_unsigned))) {
4187 if (Style.isJavaScript() &&
4188 FormatTok->isOneOf(K1: Keywords.kw_extends, K2: Keywords.kw_implements)) {
4189 JSPastExtendsOrImplements = true;
4190 // JavaScript/TypeScript supports inline object types in
4191 // extends/implements positions:
4192 // class Foo implements {bar: number} { }
4193 nextToken();
4194 if (FormatTok->is(Kind: tok::l_brace)) {
4195 tryToParseBracedList();
4196 continue;
4197 }
4198 }
4199 if (FormatTok->is(Kind: tok::l_square) && handleCppAttributes())
4200 continue;
4201 auto *Previous = FormatTok;
4202 nextToken();
4203 switch (FormatTok->Tok.getKind()) {
4204 case tok::l_paren:
4205 // We can have macros in between 'class' and the class name.
4206 if (IsJavaRecord || !IsNonMacroIdentifier(Previous) ||
4207 // e.g. `struct macro(a) S { int i; };`
4208 Previous->Previous == &InitialToken) {
4209 parseParens();
4210 }
4211 break;
4212 case tok::coloncolon:
4213 case tok::hashhash:
4214 break;
4215 default:
4216 if (JSPastExtendsOrImplements || ClassName ||
4217 Previous->isNot(Kind: tok::identifier) || Previous->is(TT: TT_AttributeMacro)) {
4218 break;
4219 }
4220 if (const auto Text = Previous->TokenText;
4221 Text.size() == 1 || Text != Text.upper()) {
4222 ClassName = Previous;
4223 }
4224 }
4225 }
4226
4227 auto IsListInitialization = [&] {
4228 if (!ClassName || IsDerived || JSPastExtendsOrImplements)
4229 return false;
4230 assert(FormatTok->is(tok::l_brace));
4231 const auto *Prev = FormatTok->getPreviousNonComment();
4232 assert(Prev);
4233 return Prev != ClassName && Prev->is(Kind: tok::identifier) &&
4234 Prev->isNot(Kind: Keywords.kw_final) && tryToParseBracedList();
4235 };
4236
4237 if (FormatTok->isOneOf(K1: tok::colon, K2: tok::less)) {
4238 int AngleNestingLevel = 0;
4239 do {
4240 if (FormatTok->is(Kind: tok::less))
4241 ++AngleNestingLevel;
4242 else if (FormatTok->is(Kind: tok::greater))
4243 --AngleNestingLevel;
4244
4245 if (AngleNestingLevel == 0) {
4246 if (FormatTok->is(Kind: tok::colon)) {
4247 IsDerived = true;
4248 } else if (!IsDerived && FormatTok->is(Kind: tok::identifier) &&
4249 FormatTok->Previous->is(Kind: tok::coloncolon)) {
4250 ClassName = FormatTok;
4251 } else if (FormatTok->is(Kind: tok::l_paren) &&
4252 IsNonMacroIdentifier(FormatTok->Previous)) {
4253 break;
4254 }
4255 }
4256 if (FormatTok->is(Kind: tok::l_brace)) {
4257 if (AngleNestingLevel == 0 && IsListInitialization())
4258 return;
4259 calculateBraceTypes(/*ExpectClassBody=*/true);
4260 if (!tryToParseBracedList())
4261 break;
4262 }
4263 if (FormatTok->is(Kind: tok::l_square)) {
4264 FormatToken *Previous = FormatTok->Previous;
4265 if (!Previous || (Previous->isNot(Kind: tok::r_paren) &&
4266 !Previous->isTypeOrIdentifier(LangOpts))) {
4267 // Don't try parsing a lambda if we had a closing parenthesis before,
4268 // it was probably a pointer to an array: int (*)[].
4269 if (!tryToParseLambda())
4270 continue;
4271 } else {
4272 parseSquare();
4273 continue;
4274 }
4275 }
4276 if (FormatTok->is(Kind: tok::semi))
4277 return;
4278 if (Style.isCSharp() && FormatTok->is(II: Keywords.kw_where)) {
4279 addUnwrappedLine();
4280 nextToken();
4281 parseCSharpGenericTypeConstraint();
4282 break;
4283 }
4284 nextToken();
4285 } while (!eof());
4286 }
4287
4288 auto GetBraceTypes =
4289 [](const FormatToken &RecordTok) -> std::pair<TokenType, TokenType> {
4290 switch (RecordTok.Tok.getKind()) {
4291 case tok::kw_class:
4292 return {TT_ClassLBrace, TT_ClassRBrace};
4293 case tok::kw_struct:
4294 return {TT_StructLBrace, TT_StructRBrace};
4295 case tok::kw_union:
4296 return {TT_UnionLBrace, TT_UnionRBrace};
4297 default:
4298 // Useful for e.g. interface.
4299 return {TT_RecordLBrace, TT_RecordRBrace};
4300 }
4301 };
4302 if (FormatTok->is(Kind: tok::l_brace)) {
4303 if (IsListInitialization())
4304 return;
4305 if (ClassName)
4306 ClassName->setFinalizedType(TT_ClassHeadName);
4307 auto [OpenBraceType, ClosingBraceType] = GetBraceTypes(InitialToken);
4308 FormatTok->setFinalizedType(OpenBraceType);
4309 if (ParseAsExpr) {
4310 parseChildBlock();
4311 } else {
4312 if (ShouldBreakBeforeBrace(Style, InitialToken,
4313 IsEmptyBlock: Tokens->peekNextToken()->is(Kind: tok::r_brace),
4314 IsJavaRecord)) {
4315 addUnwrappedLine();
4316 }
4317
4318 bool IndentAfterExplicitAccessModifier = false;
4319 unsigned AddLevels = 1u;
4320 switch (Style.IndentAccessModifiers) {
4321 case FormatStyle::IAMS_Never:
4322 break;
4323 case FormatStyle::IAMS_AfterFirstAccessModifier:
4324 if (Style.isCpp()) {
4325 IndentAfterExplicitAccessModifier = true;
4326 break;
4327 }
4328 // Other languages use the same indentation as IAMS_Always.
4329 [[fallthrough]];
4330 case FormatStyle::IAMS_Always:
4331 AddLevels = 2u;
4332 break;
4333 }
4334 parseBlock(/*MustBeDeclaration=*/true, AddLevels, /*MunchSemi=*/false,
4335 /*KeepBraces=*/true, /*IfKind=*/nullptr,
4336 /*UnindentWhitesmithsBraces=*/false,
4337 IndentAfterExplicitAccessModifier);
4338 }
4339 setPreviousRBraceType(ClosingBraceType);
4340 }
4341 // There is no addUnwrappedLine() here so that we fall through to parsing a
4342 // structural element afterwards. Thus, in "class A {} n, m;",
4343 // "} n, m;" will end up in one unwrapped line.
4344}
4345
4346void UnwrappedLineParser::parseObjCMethod() {
4347 assert(FormatTok->isOneOf(tok::l_paren, tok::identifier) &&
4348 "'(' or identifier expected.");
4349 do {
4350 if (FormatTok->is(Kind: tok::semi)) {
4351 nextToken();
4352 addUnwrappedLine();
4353 return;
4354 } else if (FormatTok->is(Kind: tok::l_brace)) {
4355 if (Style.BraceWrapping.AfterFunction)
4356 addUnwrappedLine();
4357 parseBlock();
4358 addUnwrappedLine();
4359 return;
4360 } else {
4361 nextToken();
4362 }
4363 } while (!eof());
4364}
4365
4366void UnwrappedLineParser::parseObjCProtocolList() {
4367 assert(FormatTok->is(tok::less) && "'<' expected.");
4368 do {
4369 nextToken();
4370 // Early exit in case someone forgot a close angle.
4371 if (FormatTok->isOneOf(K1: tok::semi, K2: tok::l_brace, Ks: tok::objc_end))
4372 return;
4373 } while (!eof() && FormatTok->isNot(Kind: tok::greater));
4374 nextToken(); // Skip '>'.
4375}
4376
4377void UnwrappedLineParser::parseObjCUntilAtEnd() {
4378 do {
4379 if (FormatTok->is(Kind: tok::objc_end)) {
4380 nextToken();
4381 addUnwrappedLine();
4382 break;
4383 }
4384 if (FormatTok->is(Kind: tok::l_brace)) {
4385 parseBlock();
4386 // In ObjC interfaces, nothing should be following the "}".
4387 addUnwrappedLine();
4388 } else if (FormatTok->is(Kind: tok::r_brace)) {
4389 // Ignore stray "}". parseStructuralElement doesn't consume them.
4390 nextToken();
4391 addUnwrappedLine();
4392 } else if (FormatTok->isOneOf(K1: tok::minus, K2: tok::plus)) {
4393 nextToken();
4394 if (FormatTok->isOneOf(K1: tok::l_paren, K2: tok::identifier))
4395 parseObjCMethod();
4396 } else {
4397 parseStructuralElement();
4398 }
4399 } while (!eof());
4400}
4401
4402void UnwrappedLineParser::parseObjCInterfaceOrImplementation() {
4403 assert(FormatTok->isOneOf(tok::objc_interface, tok::objc_implementation));
4404 nextToken();
4405 nextToken(); // interface name
4406
4407 // @interface can be followed by a lightweight generic
4408 // specialization list, then either a base class or a category.
4409 if (FormatTok->is(Kind: tok::less))
4410 parseObjCLightweightGenerics();
4411 if (FormatTok->is(Kind: tok::colon)) {
4412 nextToken();
4413 nextToken(); // base class name
4414 // The base class can also have lightweight generics applied to it.
4415 if (FormatTok->is(Kind: tok::less))
4416 parseObjCLightweightGenerics();
4417 } else if (FormatTok->is(Kind: tok::l_paren)) {
4418 // Skip category, if present.
4419 parseParens();
4420 }
4421
4422 if (FormatTok->is(Kind: tok::less))
4423 parseObjCProtocolList();
4424
4425 if (FormatTok->is(Kind: tok::l_brace)) {
4426 if (Style.BraceWrapping.AfterObjCDeclaration)
4427 addUnwrappedLine();
4428 parseBlock(/*MustBeDeclaration=*/true);
4429 }
4430
4431 // With instance variables, this puts '}' on its own line. Without instance
4432 // variables, this ends the @interface line.
4433 addUnwrappedLine();
4434
4435 parseObjCUntilAtEnd();
4436}
4437
4438void UnwrappedLineParser::parseObjCLightweightGenerics() {
4439 assert(FormatTok->is(tok::less));
4440 // Unlike protocol lists, generic parameterizations support
4441 // nested angles:
4442 //
4443 // @interface Foo<ValueType : id <NSCopying, NSSecureCoding>> :
4444 // NSObject <NSCopying, NSSecureCoding>
4445 //
4446 // so we need to count how many open angles we have left.
4447 unsigned NumOpenAngles = 1;
4448 do {
4449 nextToken();
4450 // Early exit in case someone forgot a close angle.
4451 if (FormatTok->isOneOf(K1: tok::semi, K2: tok::l_brace, Ks: tok::objc_end))
4452 break;
4453 if (FormatTok->is(Kind: tok::less)) {
4454 ++NumOpenAngles;
4455 } else if (FormatTok->is(Kind: tok::greater)) {
4456 assert(NumOpenAngles > 0 && "'>' makes NumOpenAngles negative");
4457 --NumOpenAngles;
4458 }
4459 } while (!eof() && NumOpenAngles != 0);
4460 nextToken(); // Skip '>'.
4461}
4462
4463// Returns true for the declaration/definition form of @protocol,
4464// false for the expression form.
4465bool UnwrappedLineParser::parseObjCProtocol() {
4466 assert(FormatTok->is(tok::objc_protocol));
4467 nextToken();
4468
4469 if (FormatTok->is(Kind: tok::l_paren)) {
4470 // The expression form of @protocol, e.g. "Protocol* p = @protocol(foo);".
4471 return false;
4472 }
4473
4474 // The definition/declaration form,
4475 // @protocol Foo
4476 // - (int)someMethod;
4477 // @end
4478
4479 nextToken(); // protocol name
4480
4481 if (FormatTok->is(Kind: tok::less))
4482 parseObjCProtocolList();
4483
4484 // Check for protocol declaration.
4485 if (FormatTok->is(Kind: tok::semi)) {
4486 nextToken();
4487 addUnwrappedLine();
4488 return true;
4489 }
4490
4491 addUnwrappedLine();
4492 parseObjCUntilAtEnd();
4493 return true;
4494}
4495
4496void UnwrappedLineParser::parseJavaScriptEs6ImportExport() {
4497 bool IsImport = FormatTok->is(II: Keywords.kw_import);
4498 assert(IsImport || FormatTok->is(tok::kw_export));
4499 nextToken();
4500
4501 // Consume the "default" in "export default class/function".
4502 if (FormatTok->is(Kind: tok::kw_default))
4503 nextToken();
4504
4505 // Consume "async function", "function" and "default function", so that these
4506 // get parsed as free-standing JS functions, i.e. do not require a trailing
4507 // semicolon.
4508 if (FormatTok->is(II: Keywords.kw_async))
4509 nextToken();
4510 if (FormatTok->is(II: Keywords.kw_function)) {
4511 nextToken();
4512 return;
4513 }
4514
4515 // For imports, `export *`, `export {...}`, consume the rest of the line up
4516 // to the terminating `;`. For everything else, just return and continue
4517 // parsing the structural element, i.e. the declaration or expression for
4518 // `export default`.
4519 if (!IsImport && FormatTok->isNoneOf(Ks: tok::l_brace, Ks: tok::star) &&
4520 !FormatTok->isStringLiteral() &&
4521 !(FormatTok->is(II: Keywords.kw_type) &&
4522 Tokens->peekNextToken()->isOneOf(K1: tok::l_brace, K2: tok::star))) {
4523 return;
4524 }
4525
4526 while (!eof()) {
4527 if (FormatTok->is(Kind: tok::semi))
4528 return;
4529 if (Line->Tokens.empty()) {
4530 // Common issue: Automatic Semicolon Insertion wrapped the line, so the
4531 // import statement should terminate.
4532 return;
4533 }
4534 if (FormatTok->is(Kind: tok::l_brace)) {
4535 FormatTok->setBlockKind(BK_Block);
4536 nextToken();
4537 parseBracedList();
4538 } else {
4539 nextToken();
4540 }
4541 }
4542}
4543
4544void UnwrappedLineParser::parseStatementMacro() {
4545 nextToken();
4546 if (FormatTok->is(Kind: tok::l_paren))
4547 parseParens();
4548 if (FormatTok->is(Kind: tok::semi))
4549 nextToken();
4550 addUnwrappedLine();
4551}
4552
4553void UnwrappedLineParser::parseVerilogHierarchyIdentifier() {
4554 // consume things like a::`b.c[d:e] or a::*
4555 while (true) {
4556 if (FormatTok->isOneOf(K1: tok::star, K2: tok::period, Ks: tok::periodstar,
4557 Ks: tok::coloncolon, Ks: tok::hash) ||
4558 Keywords.isVerilogIdentifier(Tok: *FormatTok)) {
4559 nextToken();
4560 } else if (FormatTok->is(Kind: tok::l_square)) {
4561 parseSquare();
4562 } else {
4563 break;
4564 }
4565 }
4566}
4567
4568void UnwrappedLineParser::parseVerilogSensitivityList() {
4569 if (FormatTok->isNot(Kind: tok::at))
4570 return;
4571 nextToken();
4572 // A block event expression has 2 at signs.
4573 if (FormatTok->is(Kind: tok::at))
4574 nextToken();
4575 switch (FormatTok->Tok.getKind()) {
4576 case tok::star:
4577 nextToken();
4578 break;
4579 case tok::l_paren:
4580 parseParens();
4581 break;
4582 default:
4583 parseVerilogHierarchyIdentifier();
4584 break;
4585 }
4586}
4587
4588unsigned UnwrappedLineParser::parseVerilogHierarchyHeader() {
4589 unsigned AddLevels = 0;
4590
4591 if (FormatTok->is(II: Keywords.kw_clocking)) {
4592 nextToken();
4593 if (Keywords.isVerilogIdentifier(Tok: *FormatTok))
4594 nextToken();
4595 parseVerilogSensitivityList();
4596 if (FormatTok->is(Kind: tok::semi))
4597 nextToken();
4598 } else if (FormatTok->isOneOf(K1: tok::kw_case, K2: Keywords.kw_casex,
4599 Ks: Keywords.kw_casez, Ks: Keywords.kw_randcase,
4600 Ks: Keywords.kw_randsequence)) {
4601 if (Style.IndentCaseLabels)
4602 AddLevels++;
4603 nextToken();
4604 if (FormatTok->is(Kind: tok::l_paren)) {
4605 FormatTok->setFinalizedType(TT_ConditionLParen);
4606 parseParens();
4607 }
4608 if (FormatTok->isOneOf(K1: Keywords.kw_inside, K2: Keywords.kw_matches))
4609 nextToken();
4610 // The case header has no semicolon.
4611 } else {
4612 // "module" etc.
4613 nextToken();
4614 // all the words like the name of the module and specifiers like
4615 // "automatic" and the width of function return type
4616 while (true) {
4617 if (FormatTok->is(Kind: tok::l_square)) {
4618 auto Prev = FormatTok->getPreviousNonComment();
4619 if (Prev && Keywords.isVerilogIdentifier(Tok: *Prev))
4620 Prev->setFinalizedType(TT_VerilogDimensionedTypeName);
4621 parseSquare();
4622 } else if (Keywords.isVerilogIdentifier(Tok: *FormatTok) ||
4623 FormatTok->isOneOf(K1: tok::hash, K2: tok::hashhash, Ks: tok::coloncolon,
4624 Ks: Keywords.kw_automatic, Ks: tok::kw_static)) {
4625 nextToken();
4626 } else {
4627 break;
4628 }
4629 }
4630
4631 auto NewLine = [this]() {
4632 addUnwrappedLine();
4633 Line->IsContinuation = true;
4634 };
4635
4636 // package imports
4637 while (FormatTok->is(II: Keywords.kw_import)) {
4638 NewLine();
4639 nextToken();
4640 parseVerilogHierarchyIdentifier();
4641 if (FormatTok->is(Kind: tok::semi))
4642 nextToken();
4643 }
4644
4645 // parameters and ports
4646 if (FormatTok->is(II: Keywords.kw_verilogHash)) {
4647 NewLine();
4648 nextToken();
4649 if (FormatTok->is(Kind: tok::l_paren)) {
4650 FormatTok->setFinalizedType(TT_VerilogMultiLineListLParen);
4651 parseParens();
4652 }
4653 }
4654 if (FormatTok->is(Kind: tok::l_paren)) {
4655 NewLine();
4656 FormatTok->setFinalizedType(TT_VerilogMultiLineListLParen);
4657 parseParens();
4658 }
4659
4660 // extends and implements
4661 if (FormatTok->is(II: Keywords.kw_extends)) {
4662 NewLine();
4663 nextToken();
4664 parseVerilogHierarchyIdentifier();
4665 if (FormatTok->is(Kind: tok::l_paren))
4666 parseParens();
4667 }
4668 if (FormatTok->is(II: Keywords.kw_implements)) {
4669 NewLine();
4670 do {
4671 nextToken();
4672 parseVerilogHierarchyIdentifier();
4673 } while (FormatTok->is(Kind: tok::comma));
4674 }
4675
4676 // Coverage event for cover groups.
4677 if (FormatTok->is(Kind: tok::at)) {
4678 NewLine();
4679 parseVerilogSensitivityList();
4680 }
4681
4682 if (FormatTok->is(Kind: tok::semi))
4683 nextToken(/*LevelDifference=*/1);
4684 addUnwrappedLine();
4685 }
4686
4687 return AddLevels;
4688}
4689
4690void UnwrappedLineParser::parseVerilogTable() {
4691 assert(FormatTok->is(Keywords.kw_table));
4692 nextToken(/*LevelDifference=*/1);
4693 addUnwrappedLine();
4694
4695 auto InitialLevel = Line->Level++;
4696 while (!eof() && !Keywords.isVerilogEnd(Tok: *FormatTok)) {
4697 FormatToken *Tok = FormatTok;
4698 nextToken();
4699 if (Tok->is(Kind: tok::semi))
4700 addUnwrappedLine();
4701 else if (Tok->isOneOf(K1: tok::star, K2: tok::colon, Ks: tok::question, Ks: tok::minus))
4702 Tok->setFinalizedType(TT_VerilogTableItem);
4703 }
4704 Line->Level = InitialLevel;
4705 nextToken(/*LevelDifference=*/-1);
4706 addUnwrappedLine();
4707}
4708
4709void UnwrappedLineParser::parseVerilogCaseLabel() {
4710 // The label will get unindented in AnnotatingParser. If there are no leading
4711 // spaces, indent the rest here so that things inside the block will be
4712 // indented relative to things outside. We don't use parseLabel because we
4713 // don't know whether this colon is a label or a ternary expression at this
4714 // point.
4715 auto OrigLevel = Line->Level;
4716 auto FirstLine = CurrentLines->size();
4717 if (Line->Level == 0 || (Line->InPPDirective && Line->Level <= 1))
4718 ++Line->Level;
4719 else if (!Style.IndentCaseBlocks && Keywords.isVerilogBegin(Tok: *FormatTok))
4720 --Line->Level;
4721 parseStructuralElement();
4722 // Restore the indentation in both the new line and the line that has the
4723 // label.
4724 if (CurrentLines->size() > FirstLine)
4725 (*CurrentLines)[FirstLine].Level = OrigLevel;
4726 Line->Level = OrigLevel;
4727}
4728
4729void UnwrappedLineParser::parseVerilogExtern() {
4730 assert(
4731 FormatTok->isOneOf(tok::kw_extern, tok::kw_export, Keywords.kw_import));
4732 nextToken();
4733 // "DPI-C"
4734 if (FormatTok->is(Kind: tok::string_literal))
4735 nextToken();
4736 skipVerilogQualifiers();
4737 if (Keywords.isVerilogIdentifier(Tok: *FormatTok))
4738 nextToken();
4739 if (FormatTok->is(Kind: tok::equal))
4740 nextToken();
4741 if (Keywords.isVerilogHierarchy(Tok: *FormatTok))
4742 parseVerilogHierarchyHeader();
4743}
4744
4745void UnwrappedLineParser::skipVerilogQualifiers() {
4746 while (FormatTok->isOneOf(K1: tok::kw_protected, K2: tok::kw_virtual, Ks: tok::kw_static,
4747 Ks: Keywords.kw_rand, Ks: Keywords.kw_context,
4748 Ks: Keywords.kw_pure, Ks: Keywords.kw_randc,
4749 Ks: Keywords.kw_local)) {
4750 nextToken();
4751 }
4752}
4753
4754bool UnwrappedLineParser::containsExpansion(const UnwrappedLine &Line) const {
4755 for (const auto &N : Line.Tokens) {
4756 if (N.Tok->MacroCtx)
4757 return true;
4758 for (const UnwrappedLine &Child : N.Children)
4759 if (containsExpansion(Line: Child))
4760 return true;
4761 }
4762 return false;
4763}
4764
4765void UnwrappedLineParser::addUnwrappedLine(LineLevel AdjustLevel) {
4766 if (Line->Tokens.empty())
4767 return;
4768 LLVM_DEBUG({
4769 if (!parsingPPDirective()) {
4770 llvm::dbgs() << "Adding unwrapped line:\n";
4771 printDebugInfo(*Line);
4772 }
4773 });
4774
4775 // If this line closes a block when in Whitesmiths mode, remember that
4776 // information so that the level can be decreased after the line is added.
4777 // This has to happen after the addition of the line since the line itself
4778 // needs to be indented.
4779 bool ClosesWhitesmithsBlock =
4780 Line->MatchingOpeningBlockLineIndex != UnwrappedLine::kInvalidIndex &&
4781 Style.BreakBeforeBraces == FormatStyle::BS_Whitesmiths;
4782
4783 // If the current line was expanded from a macro call, we use it to
4784 // reconstruct an unwrapped line from the structure of the expanded unwrapped
4785 // line and the unexpanded token stream.
4786 if (!parsingPPDirective() && !InExpansion && containsExpansion(Line: *Line)) {
4787 if (!Reconstruct)
4788 Reconstruct.emplace(args&: Line->Level, args&: Unexpanded);
4789 Reconstruct->addLine(Line: *Line);
4790
4791 // While the reconstructed unexpanded lines are stored in the normal
4792 // flow of lines, the expanded lines are stored on the side to be analyzed
4793 // in an extra step.
4794 CurrentExpandedLines.push_back(Elt: std::move(*Line));
4795
4796 if (Reconstruct->finished()) {
4797 UnwrappedLine Reconstructed = std::move(*Reconstruct).takeResult();
4798 assert(!Reconstructed.Tokens.empty() &&
4799 "Reconstructed must at least contain the macro identifier.");
4800 assert(!parsingPPDirective());
4801 LLVM_DEBUG({
4802 llvm::dbgs() << "Adding unexpanded line:\n";
4803 printDebugInfo(Reconstructed);
4804 });
4805 ExpandedLines[Reconstructed.Tokens.begin()->Tok] = CurrentExpandedLines;
4806 Lines.push_back(Elt: std::move(Reconstructed));
4807 CurrentExpandedLines.clear();
4808 Reconstruct.reset();
4809 }
4810 } else {
4811 // At the top level we only get here when no unexpansion is going on, or
4812 // when conditional formatting led to unfinished macro reconstructions.
4813 assert(!Reconstruct || (CurrentLines != &Lines) || !PP.Stack.empty());
4814 CurrentLines->push_back(Elt: std::move(*Line));
4815 }
4816 Line->Tokens.clear();
4817 Line->MatchingOpeningBlockLineIndex = UnwrappedLine::kInvalidIndex;
4818 Line->FirstStartColumn = 0;
4819 Line->IsContinuation = false;
4820 Line->SeenDecltypeAuto = false;
4821 Line->IsModuleOrImportDecl = false;
4822
4823 if (ClosesWhitesmithsBlock && AdjustLevel == LineLevel::Remove)
4824 --Line->Level;
4825 if (!parsingPPDirective() && !PreprocessorDirectives.empty()) {
4826 CurrentLines->append(
4827 in_start: std::make_move_iterator(i: PreprocessorDirectives.begin()),
4828 in_end: std::make_move_iterator(i: PreprocessorDirectives.end()));
4829 PreprocessorDirectives.clear();
4830 }
4831 // Disconnect the current token from the last token on the previous line.
4832 FormatTok->Previous = nullptr;
4833}
4834
4835bool UnwrappedLineParser::eof() const { return FormatTok->is(Kind: tok::eof); }
4836
4837bool UnwrappedLineParser::isOnNewLine(const FormatToken &FormatTok) {
4838 return (Line->InPPDirective || FormatTok.HasUnescapedNewline) &&
4839 FormatTok.NewlinesBefore > 0;
4840}
4841
4842// Checks if \p FormatTok is a line comment that continues the line comment
4843// section on \p Line.
4844static bool
4845continuesLineCommentSection(const FormatToken &FormatTok,
4846 const UnwrappedLine &Line, const FormatStyle &Style,
4847 const llvm::Regex &CommentPragmasRegex) {
4848 if (Line.Tokens.empty() || Style.ReflowComments != FormatStyle::RCS_Always)
4849 return false;
4850
4851 StringRef IndentContent = FormatTok.TokenText;
4852 if (FormatTok.TokenText.starts_with(Prefix: "//") ||
4853 FormatTok.TokenText.starts_with(Prefix: "/*")) {
4854 IndentContent = FormatTok.TokenText.substr(Start: 2);
4855 }
4856 if (CommentPragmasRegex.match(String: IndentContent))
4857 return false;
4858
4859 // If Line starts with a line comment, then FormatTok continues the comment
4860 // section if its original column is greater or equal to the original start
4861 // column of the line.
4862 //
4863 // Define the min column token of a line as follows: if a line ends in '{' or
4864 // contains a '{' followed by a line comment, then the min column token is
4865 // that '{'. Otherwise, the min column token of the line is the first token of
4866 // the line.
4867 //
4868 // If Line starts with a token other than a line comment, then FormatTok
4869 // continues the comment section if its original column is greater than the
4870 // original start column of the min column token of the line.
4871 //
4872 // For example, the second line comment continues the first in these cases:
4873 //
4874 // // first line
4875 // // second line
4876 //
4877 // and:
4878 //
4879 // // first line
4880 // // second line
4881 //
4882 // and:
4883 //
4884 // int i; // first line
4885 // // second line
4886 //
4887 // and:
4888 //
4889 // do { // first line
4890 // // second line
4891 // int i;
4892 // } while (true);
4893 //
4894 // and:
4895 //
4896 // enum {
4897 // a, // first line
4898 // // second line
4899 // b
4900 // };
4901 //
4902 // The second line comment doesn't continue the first in these cases:
4903 //
4904 // // first line
4905 // // second line
4906 //
4907 // and:
4908 //
4909 // int i; // first line
4910 // // second line
4911 //
4912 // and:
4913 //
4914 // do { // first line
4915 // // second line
4916 // int i;
4917 // } while (true);
4918 //
4919 // and:
4920 //
4921 // enum {
4922 // a, // first line
4923 // // second line
4924 // };
4925 const FormatToken *MinColumnToken = Line.Tokens.front().Tok;
4926
4927 // Scan for '{//'. If found, use the column of '{' as a min column for line
4928 // comment section continuation.
4929 const FormatToken *PreviousToken = nullptr;
4930 for (const UnwrappedLineNode &Node : Line.Tokens) {
4931 if (PreviousToken && PreviousToken->is(Kind: tok::l_brace) &&
4932 isLineComment(FormatTok: *Node.Tok)) {
4933 MinColumnToken = PreviousToken;
4934 break;
4935 }
4936 PreviousToken = Node.Tok;
4937
4938 // Grab the last newline preceding a token in this unwrapped line.
4939 if (Node.Tok->NewlinesBefore > 0)
4940 MinColumnToken = Node.Tok;
4941 }
4942 if (PreviousToken && PreviousToken->is(Kind: tok::l_brace))
4943 MinColumnToken = PreviousToken;
4944
4945 return continuesLineComment(FormatTok, /*Previous=*/Line.Tokens.back().Tok,
4946 MinColumnToken);
4947}
4948
4949void UnwrappedLineParser::flushComments(bool NewlineBeforeNext) {
4950 bool JustComments = Line->Tokens.empty();
4951 for (FormatToken *Tok : CommentsBeforeNextToken) {
4952 // Line comments that belong to the same line comment section are put on the
4953 // same line since later we might want to reflow content between them.
4954 // Additional fine-grained breaking of line comment sections is controlled
4955 // by the class BreakableLineCommentSection in case it is desirable to keep
4956 // several line comment sections in the same unwrapped line.
4957 //
4958 // FIXME: Consider putting separate line comment sections as children to the
4959 // unwrapped line instead.
4960 Tok->ContinuesLineCommentSection =
4961 continuesLineCommentSection(FormatTok: *Tok, Line: *Line, Style, CommentPragmasRegex);
4962 if (isOnNewLine(FormatTok: *Tok) && JustComments && !Tok->ContinuesLineCommentSection)
4963 addUnwrappedLine();
4964 pushToken(Tok);
4965 }
4966 if (NewlineBeforeNext && JustComments)
4967 addUnwrappedLine();
4968 CommentsBeforeNextToken.clear();
4969}
4970
4971void UnwrappedLineParser::nextToken(int LevelDifference) {
4972 if (eof())
4973 return;
4974 flushComments(NewlineBeforeNext: isOnNewLine(FormatTok: *FormatTok));
4975 pushToken(Tok: FormatTok);
4976 FormatToken *Previous = FormatTok;
4977 if (!Style.isJavaScript())
4978 readToken(LevelDifference);
4979 else
4980 readTokenWithJavaScriptASI();
4981 FormatTok->Previous = Previous;
4982 if (Style.isVerilog()) {
4983 // Blocks in Verilog can have `begin` and `end` instead of braces. For
4984 // keywords like `begin`, we can't treat them the same as left braces
4985 // because some contexts require one of them. For example structs use
4986 // braces and if blocks use keywords, and a left brace can occur in an if
4987 // statement, but it is not a block. For keywords like `end`, we simply
4988 // treat them the same as right braces.
4989 if (Keywords.isVerilogEnd(Tok: *FormatTok))
4990 FormatTok->Tok.setKind(tok::r_brace);
4991 }
4992}
4993
4994void UnwrappedLineParser::distributeComments(
4995 const ArrayRef<FormatToken *> &Comments, const FormatToken *NextTok) {
4996 // Whether or not a line comment token continues a line is controlled by
4997 // the method continuesLineCommentSection, with the following caveat:
4998 //
4999 // Define a trail of Comments to be a nonempty proper postfix of Comments such
5000 // that each comment line from the trail is aligned with the next token, if
5001 // the next token exists. If a trail exists, the beginning of the maximal
5002 // trail is marked as a start of a new comment section.
5003 //
5004 // For example in this code:
5005 //
5006 // int a; // line about a
5007 // // line 1 about b
5008 // // line 2 about b
5009 // int b;
5010 //
5011 // the two lines about b form a maximal trail, so there are two sections, the
5012 // first one consisting of the single comment "// line about a" and the
5013 // second one consisting of the next two comments.
5014 if (Comments.empty())
5015 return;
5016 bool ShouldPushCommentsInCurrentLine = true;
5017 bool HasTrailAlignedWithNextToken = false;
5018 unsigned StartOfTrailAlignedWithNextToken = 0;
5019 if (NextTok) {
5020 // We are skipping the first element intentionally.
5021 for (unsigned i = Comments.size() - 1; i > 0; --i) {
5022 if (Comments[i]->OriginalColumn == NextTok->OriginalColumn) {
5023 HasTrailAlignedWithNextToken = true;
5024 StartOfTrailAlignedWithNextToken = i;
5025 }
5026 }
5027 }
5028 for (unsigned i = 0, e = Comments.size(); i < e; ++i) {
5029 FormatToken *FormatTok = Comments[i];
5030 if (HasTrailAlignedWithNextToken && i == StartOfTrailAlignedWithNextToken) {
5031 FormatTok->ContinuesLineCommentSection = false;
5032 } else {
5033 FormatTok->ContinuesLineCommentSection = continuesLineCommentSection(
5034 FormatTok: *FormatTok, Line: *Line, Style, CommentPragmasRegex);
5035 }
5036 if (!FormatTok->ContinuesLineCommentSection &&
5037 (isOnNewLine(FormatTok: *FormatTok) || FormatTok->IsFirst)) {
5038 ShouldPushCommentsInCurrentLine = false;
5039 }
5040 if (ShouldPushCommentsInCurrentLine)
5041 pushToken(Tok: FormatTok);
5042 else
5043 CommentsBeforeNextToken.push_back(Elt: FormatTok);
5044 }
5045}
5046
5047void UnwrappedLineParser::readToken(int LevelDifference) {
5048 SmallVector<FormatToken *, 1> Comments;
5049 bool PreviousWasComment = false;
5050 bool FirstNonCommentOnLine = false;
5051 do {
5052 FormatTok = Tokens->getNextToken();
5053 assert(FormatTok);
5054 while (FormatTok->isOneOf(K1: TT_ConflictStart, K2: TT_ConflictEnd,
5055 Ks: TT_ConflictAlternative)) {
5056 if (FormatTok->is(TT: TT_ConflictStart))
5057 conditionalCompilationStart(/*Unreachable=*/false);
5058 else if (FormatTok->is(TT: TT_ConflictAlternative))
5059 conditionalCompilationAlternative();
5060 else if (FormatTok->is(TT: TT_ConflictEnd))
5061 conditionalCompilationEnd();
5062 FormatTok = Tokens->getNextToken();
5063 FormatTok->MustBreakBefore = true;
5064 FormatTok->MustBreakBeforeFinalized = true;
5065 }
5066
5067 auto IsFirstNonCommentOnLine = [](bool FirstNonCommentOnLine,
5068 const FormatToken &Tok,
5069 bool PreviousWasComment) {
5070 auto IsFirstOnLine = [](const FormatToken &Tok) {
5071 return Tok.HasUnescapedNewline || Tok.IsFirst;
5072 };
5073
5074 // Consider preprocessor directives preceded by block comments as first
5075 // on line.
5076 if (PreviousWasComment)
5077 return FirstNonCommentOnLine || IsFirstOnLine(Tok);
5078 return IsFirstOnLine(Tok);
5079 };
5080
5081 FirstNonCommentOnLine = IsFirstNonCommentOnLine(
5082 FirstNonCommentOnLine, *FormatTok, PreviousWasComment);
5083 PreviousWasComment = FormatTok->is(Kind: tok::comment);
5084
5085 while (!Line->InPPDirective && FormatTok->is(Kind: tok::hash) &&
5086 FirstNonCommentOnLine) {
5087 // In Verilog, the backtick is used for macro invocations. In TableGen,
5088 // the single hash is used for the paste operator.
5089 const auto *Next = Tokens->peekNextToken();
5090 if ((Style.isVerilog() && !Keywords.isVerilogPPDirective(Tok: *Next)) ||
5091 (Style.isTableGen() &&
5092 Next->isNoneOf(Ks: tok::kw_else, Ks: tok::pp_define, Ks: tok::pp_ifdef,
5093 Ks: tok::pp_ifndef, Ks: tok::pp_endif))) {
5094 break;
5095 }
5096 distributeComments(Comments, NextTok: FormatTok);
5097 Comments.clear();
5098 // If the directive was parsed before the token stream was rewound (see
5099 // parseMacroCall()), its lines were kept. Parse it again only for its
5100 // effect on the preprocessor bookkeeping and discard the new lines.
5101 const bool ParsedBefore = !ParsedPPDirectives.insert(Ptr: FormatTok).second;
5102 // If there is an unfinished unwrapped line, we flush the preprocessor
5103 // directives only after that unwrapped line was finished later.
5104 bool SwitchToPreprocessorLines = !Line->Tokens.empty();
5105 ScopedLineState BlockState(*this, SwitchToPreprocessorLines,
5106 /*DiscardLines=*/ParsedBefore);
5107 assert((LevelDifference >= 0 ||
5108 static_cast<unsigned>(-LevelDifference) <= Line->Level) &&
5109 "LevelDifference makes Line->Level negative");
5110 Line->Level += LevelDifference;
5111 // Comments stored before the preprocessor directive need to be output
5112 // before the preprocessor directive, at the same level as the
5113 // preprocessor directive, as we consider them to apply to the directive.
5114 if (Style.IndentPPDirectives == FormatStyle::PPDIS_BeforeHash &&
5115 PP.BranchLevel > 0) {
5116 Line->Level += PP.BranchLevel;
5117 }
5118 assert(Line->Level >= Line->UnbracedBodyLevel);
5119 Line->Level -= Line->UnbracedBodyLevel;
5120 flushComments(NewlineBeforeNext: isOnNewLine(FormatTok: *FormatTok));
5121 const bool IsEndIf = Tokens->peekNextToken()->is(Kind: tok::pp_endif);
5122 parsePPDirective();
5123 PreviousWasComment = FormatTok->is(Kind: tok::comment);
5124 FirstNonCommentOnLine = IsFirstNonCommentOnLine(
5125 FirstNonCommentOnLine, *FormatTok, PreviousWasComment);
5126 // If the #endif of a potential include guard is the last thing in the
5127 // file, then we found an include guard.
5128 if (IsEndIf && PP.IncludeGuard == IG_Defined && PP.BranchLevel == -1 &&
5129 getIncludeGuardState(Style: Style.IndentPPDirectives) == IG_Inited &&
5130 (eof() ||
5131 (PreviousWasComment &&
5132 Tokens->peekNextToken(/*SkipComment=*/true)->is(Kind: tok::eof)))) {
5133 PP.IncludeGuard = IG_Found;
5134 }
5135 }
5136
5137 if (!PP.Stack.empty() && (PP.Stack.back().Kind == PP_Unreachable) &&
5138 !Line->InPPDirective) {
5139 continue;
5140 }
5141
5142 if (FormatTok->is(Kind: tok::identifier) &&
5143 Macros.defined(Name: FormatTok->TokenText) &&
5144 // FIXME: Allow expanding macros in preprocessor directives.
5145 !Line->InPPDirective) {
5146 FormatToken *ID = FormatTok;
5147 unsigned Position = Tokens->getPosition();
5148 // Parsing the arguments of the call may parse preprocessor directives,
5149 // which are parsed again if the token stream is rewound because the
5150 // arguments are discarded. The preprocessor bookkeeping is restored
5151 // whenever that happens.
5152 const auto SavedPPState = PP;
5153
5154 // To correctly parse the code, we need to replace the tokens of the macro
5155 // call with its expansion.
5156 auto PreCall = std::move(Line);
5157 Line.reset(p: new UnwrappedLine);
5158 bool OldInExpansion = InExpansion;
5159 InExpansion = true;
5160 // We parse the macro call into a new line.
5161 auto Args = parseMacroCall(SavedPPState);
5162 InExpansion = OldInExpansion;
5163 assert(Line->Tokens.front().Tok == ID);
5164 // And remember the unexpanded macro call tokens.
5165 auto UnexpandedLine = std::move(Line);
5166 // Reset to the old line.
5167 Line = std::move(PreCall);
5168
5169 LLVM_DEBUG({
5170 llvm::dbgs() << "Macro call: " << ID->TokenText << "(";
5171 if (Args) {
5172 llvm::dbgs() << "(";
5173 for (const auto &Arg : Args.value())
5174 for (const auto &T : Arg)
5175 llvm::dbgs() << T->TokenText << " ";
5176 llvm::dbgs() << ")";
5177 }
5178 llvm::dbgs() << "\n";
5179 });
5180 if (Macros.objectLike(Name: ID->TokenText) && Args &&
5181 !Macros.hasArity(Name: ID->TokenText, Arity: Args->size())) {
5182 // The macro is either
5183 // - object-like, but we got argumnets, or
5184 // - overloaded to be both object-like and function-like, but none of
5185 // the function-like arities match the number of arguments.
5186 // Thus, expand as object-like macro.
5187 LLVM_DEBUG(llvm::dbgs()
5188 << "Macro \"" << ID->TokenText
5189 << "\" not overloaded for arity " << Args->size()
5190 << "or not function-like, using object-like overload.");
5191 Args.reset();
5192 UnexpandedLine->Tokens.resize(new_size: 1);
5193 Tokens->setPosition(Position);
5194 // Not nextToken(), which would push the stale FormatTok onto the line.
5195 FormatTok = Tokens->getNextToken();
5196 PP = SavedPPState;
5197 assert(!Args && Macros.objectLike(ID->TokenText));
5198 }
5199 if ((!Args && Macros.objectLike(Name: ID->TokenText)) ||
5200 (Args && Macros.hasArity(Name: ID->TokenText, Arity: Args->size()))) {
5201 // Next, we insert the expanded tokens in the token stream at the
5202 // current position, and continue parsing.
5203 Unexpanded[ID] = std::move(UnexpandedLine);
5204 SmallVector<FormatToken *, 8> Expansion =
5205 Macros.expand(ID, OptionalArgs: std::move(Args));
5206 if (!Expansion.empty())
5207 FormatTok = Tokens->insertTokens(Tokens: Expansion);
5208
5209 LLVM_DEBUG({
5210 llvm::dbgs() << "Expanded: ";
5211 for (const auto &T : Expansion)
5212 llvm::dbgs() << T->TokenText << " ";
5213 llvm::dbgs() << "\n";
5214 });
5215 } else {
5216 LLVM_DEBUG({
5217 llvm::dbgs() << "Did not expand macro \"" << ID->TokenText
5218 << "\", because it was used ";
5219 if (Args)
5220 llvm::dbgs() << "with " << Args->size();
5221 else
5222 llvm::dbgs() << "without";
5223 llvm::dbgs() << " arguments, which doesn't match any definition.\n";
5224 });
5225 Tokens->setPosition(Position);
5226 FormatTok = ID;
5227 PP = SavedPPState;
5228 }
5229 }
5230
5231 if (FormatTok->isNot(Kind: tok::comment)) {
5232 distributeComments(Comments, NextTok: FormatTok);
5233 Comments.clear();
5234 return;
5235 }
5236
5237 Comments.push_back(Elt: FormatTok);
5238 } while (!eof());
5239
5240 distributeComments(Comments, NextTok: nullptr);
5241 Comments.clear();
5242}
5243
5244namespace {
5245template <typename Iterator>
5246void pushTokens(Iterator Begin, Iterator End,
5247 SmallVectorImpl<FormatToken *> &Into) {
5248 for (auto I = Begin; I != End; ++I) {
5249 Into.push_back(Elt: I->Tok);
5250 for (const auto &Child : I->Children)
5251 pushTokens(Child.Tokens.begin(), Child.Tokens.end(), Into);
5252 }
5253}
5254} // namespace
5255
5256std::optional<llvm::SmallVector<llvm::SmallVector<FormatToken *, 8>, 1>>
5257UnwrappedLineParser::parseMacroCall(const PPState &SavedPPState) {
5258 std::optional<llvm::SmallVector<llvm::SmallVector<FormatToken *, 8>, 1>> Args;
5259 assert(Line->Tokens.empty());
5260 // Not nextToken(), which would already expand a directly following macro
5261 // call before the expansion of this one is inserted.
5262 auto ConsumeLastTokenOfCall = [this] {
5263 flushComments(NewlineBeforeNext: isOnNewLine(FormatTok: *FormatTok));
5264 pushToken(Tok: FormatTok);
5265 FormatTok = Tokens->getNextToken();
5266 };
5267 if (Tokens->peekNextToken(/*SkipComment=*/true)->isNot(Kind: tok::l_paren)) {
5268 ConsumeLastTokenOfCall();
5269 return Args;
5270 }
5271 nextToken();
5272 assert(FormatTok->is(tok::l_paren));
5273 unsigned Position = Tokens->getPosition();
5274 FormatToken *Tok = FormatTok;
5275 nextToken();
5276 Args.emplace();
5277 auto ArgStart = std::prev(x: Line->Tokens.end());
5278
5279 int Parens = 0;
5280 do {
5281 switch (FormatTok->Tok.getKind()) {
5282 case tok::l_paren:
5283 ++Parens;
5284 nextToken();
5285 break;
5286 case tok::r_paren: {
5287 if (Parens > 0) {
5288 --Parens;
5289 nextToken();
5290 break;
5291 }
5292 Args->push_back(Elt: {});
5293 pushTokens(Begin: std::next(x: ArgStart), End: Line->Tokens.end(), Into&: Args->back());
5294 ConsumeLastTokenOfCall();
5295 return Args;
5296 }
5297 case tok::comma: {
5298 if (Parens > 0) {
5299 nextToken();
5300 break;
5301 }
5302 Args->push_back(Elt: {});
5303 pushTokens(Begin: std::next(x: ArgStart), End: Line->Tokens.end(), Into&: Args->back());
5304 nextToken();
5305 ArgStart = std::prev(x: Line->Tokens.end());
5306 break;
5307 }
5308 default:
5309 nextToken();
5310 break;
5311 }
5312 } while (!eof());
5313 Line->Tokens.resize(new_size: 1);
5314 Tokens->setPosition(Position);
5315 FormatTok = Tok;
5316 PP = SavedPPState;
5317 return {};
5318}
5319
5320void UnwrappedLineParser::pushToken(FormatToken *Tok) {
5321 Line->Tokens.push_back(x: UnwrappedLineNode(Tok));
5322 if (PP.AtEndOfPPLine) {
5323 auto &Tok = *Line->Tokens.back().Tok;
5324 Tok.MustBreakBefore = true;
5325 Tok.MustBreakBeforeFinalized = true;
5326 Tok.FirstAfterPPLine = true;
5327 PP.AtEndOfPPLine = false;
5328 }
5329}
5330
5331} // end namespace format
5332} // end namespace clang
5333