1 //===--- Format.cpp - Format C++ code -------------------------------------===//
2 //
3 //                     The LLVM Compiler Infrastructure
4 //
5 // This file is distributed under the University of Illinois Open Source
6 // License. See LICENSE.TXT for details.
7 //
8 //===----------------------------------------------------------------------===//
9 ///
10 /// \file
11 /// \brief This file implements functions declared in Format.h. This will be
12 /// split into separate files as we go.
13 ///
14 //===----------------------------------------------------------------------===//
15 
16 #include "ContinuationIndenter.h"
17 #include "TokenAnnotator.h"
18 #include "UnwrappedLineFormatter.h"
19 #include "UnwrappedLineParser.h"
20 #include "WhitespaceManager.h"
21 #include "clang/Basic/Diagnostic.h"
22 #include "clang/Basic/DiagnosticOptions.h"
23 #include "clang/Basic/SourceManager.h"
24 #include "clang/Format/Format.h"
25 #include "clang/Lex/Lexer.h"
26 #include "llvm/ADT/STLExtras.h"
27 #include "llvm/Support/Allocator.h"
28 #include "llvm/Support/Debug.h"
29 #include "llvm/Support/Path.h"
30 #include "llvm/Support/YAMLTraits.h"
31 #include <queue>
32 #include <string>
33 
34 #define DEBUG_TYPE "format-formatter"
35 
36 using clang::format::FormatStyle;
37 
38 LLVM_YAML_IS_FLOW_SEQUENCE_VECTOR(std::string)
39 
40 namespace llvm {
41 namespace yaml {
42 template <> struct ScalarEnumerationTraits<FormatStyle::LanguageKind> {
43   static void enumeration(IO &IO, FormatStyle::LanguageKind &Value) {
44     IO.enumCase(Value, "Cpp", FormatStyle::LK_Cpp);
45     IO.enumCase(Value, "Java", FormatStyle::LK_Java);
46     IO.enumCase(Value, "JavaScript", FormatStyle::LK_JavaScript);
47     IO.enumCase(Value, "Proto", FormatStyle::LK_Proto);
48   }
49 };
50 
51 template <> struct ScalarEnumerationTraits<FormatStyle::LanguageStandard> {
52   static void enumeration(IO &IO, FormatStyle::LanguageStandard &Value) {
53     IO.enumCase(Value, "Cpp03", FormatStyle::LS_Cpp03);
54     IO.enumCase(Value, "C++03", FormatStyle::LS_Cpp03);
55     IO.enumCase(Value, "Cpp11", FormatStyle::LS_Cpp11);
56     IO.enumCase(Value, "C++11", FormatStyle::LS_Cpp11);
57     IO.enumCase(Value, "Auto", FormatStyle::LS_Auto);
58   }
59 };
60 
61 template <> struct ScalarEnumerationTraits<FormatStyle::UseTabStyle> {
62   static void enumeration(IO &IO, FormatStyle::UseTabStyle &Value) {
63     IO.enumCase(Value, "Never", FormatStyle::UT_Never);
64     IO.enumCase(Value, "false", FormatStyle::UT_Never);
65     IO.enumCase(Value, "Always", FormatStyle::UT_Always);
66     IO.enumCase(Value, "true", FormatStyle::UT_Always);
67     IO.enumCase(Value, "ForIndentation", FormatStyle::UT_ForIndentation);
68   }
69 };
70 
71 template <> struct ScalarEnumerationTraits<FormatStyle::ShortFunctionStyle> {
72   static void enumeration(IO &IO, FormatStyle::ShortFunctionStyle &Value) {
73     IO.enumCase(Value, "None", FormatStyle::SFS_None);
74     IO.enumCase(Value, "false", FormatStyle::SFS_None);
75     IO.enumCase(Value, "All", FormatStyle::SFS_All);
76     IO.enumCase(Value, "true", FormatStyle::SFS_All);
77     IO.enumCase(Value, "Inline", FormatStyle::SFS_Inline);
78     IO.enumCase(Value, "Empty", FormatStyle::SFS_Empty);
79   }
80 };
81 
82 template <> struct ScalarEnumerationTraits<FormatStyle::BinaryOperatorStyle> {
83   static void enumeration(IO &IO, FormatStyle::BinaryOperatorStyle &Value) {
84     IO.enumCase(Value, "All", FormatStyle::BOS_All);
85     IO.enumCase(Value, "true", FormatStyle::BOS_All);
86     IO.enumCase(Value, "None", FormatStyle::BOS_None);
87     IO.enumCase(Value, "false", FormatStyle::BOS_None);
88     IO.enumCase(Value, "NonAssignment", FormatStyle::BOS_NonAssignment);
89   }
90 };
91 
92 template <> struct ScalarEnumerationTraits<FormatStyle::BraceBreakingStyle> {
93   static void enumeration(IO &IO, FormatStyle::BraceBreakingStyle &Value) {
94     IO.enumCase(Value, "Attach", FormatStyle::BS_Attach);
95     IO.enumCase(Value, "Linux", FormatStyle::BS_Linux);
96     IO.enumCase(Value, "Stroustrup", FormatStyle::BS_Stroustrup);
97     IO.enumCase(Value, "Allman", FormatStyle::BS_Allman);
98     IO.enumCase(Value, "GNU", FormatStyle::BS_GNU);
99   }
100 };
101 
102 template <>
103 struct ScalarEnumerationTraits<FormatStyle::NamespaceIndentationKind> {
104   static void enumeration(IO &IO,
105                           FormatStyle::NamespaceIndentationKind &Value) {
106     IO.enumCase(Value, "None", FormatStyle::NI_None);
107     IO.enumCase(Value, "Inner", FormatStyle::NI_Inner);
108     IO.enumCase(Value, "All", FormatStyle::NI_All);
109   }
110 };
111 
112 template <>
113 struct ScalarEnumerationTraits<FormatStyle::PointerAlignmentStyle> {
114   static void enumeration(IO &IO,
115                           FormatStyle::PointerAlignmentStyle &Value) {
116     IO.enumCase(Value, "Middle", FormatStyle::PAS_Middle);
117     IO.enumCase(Value, "Left", FormatStyle::PAS_Left);
118     IO.enumCase(Value, "Right", FormatStyle::PAS_Right);
119 
120     // For backward compatibility.
121     IO.enumCase(Value, "true", FormatStyle::PAS_Left);
122     IO.enumCase(Value, "false", FormatStyle::PAS_Right);
123   }
124 };
125 
126 template <>
127 struct ScalarEnumerationTraits<FormatStyle::SpaceBeforeParensOptions> {
128   static void enumeration(IO &IO,
129                           FormatStyle::SpaceBeforeParensOptions &Value) {
130     IO.enumCase(Value, "Never", FormatStyle::SBPO_Never);
131     IO.enumCase(Value, "ControlStatements",
132                 FormatStyle::SBPO_ControlStatements);
133     IO.enumCase(Value, "Always", FormatStyle::SBPO_Always);
134 
135     // For backward compatibility.
136     IO.enumCase(Value, "false", FormatStyle::SBPO_Never);
137     IO.enumCase(Value, "true", FormatStyle::SBPO_ControlStatements);
138   }
139 };
140 
141 template <> struct MappingTraits<FormatStyle> {
142   static void mapping(IO &IO, FormatStyle &Style) {
143     // When reading, read the language first, we need it for getPredefinedStyle.
144     IO.mapOptional("Language", Style.Language);
145 
146     if (IO.outputting()) {
147       StringRef StylesArray[] = { "LLVM",    "Google", "Chromium",
148                                   "Mozilla", "WebKit", "GNU" };
149       ArrayRef<StringRef> Styles(StylesArray);
150       for (size_t i = 0, e = Styles.size(); i < e; ++i) {
151         StringRef StyleName(Styles[i]);
152         FormatStyle PredefinedStyle;
153         if (getPredefinedStyle(StyleName, Style.Language, &PredefinedStyle) &&
154             Style == PredefinedStyle) {
155           IO.mapOptional("# BasedOnStyle", StyleName);
156           break;
157         }
158       }
159     } else {
160       StringRef BasedOnStyle;
161       IO.mapOptional("BasedOnStyle", BasedOnStyle);
162       if (!BasedOnStyle.empty()) {
163         FormatStyle::LanguageKind OldLanguage = Style.Language;
164         FormatStyle::LanguageKind Language =
165             ((FormatStyle *)IO.getContext())->Language;
166         if (!getPredefinedStyle(BasedOnStyle, Language, &Style)) {
167           IO.setError(Twine("Unknown value for BasedOnStyle: ", BasedOnStyle));
168           return;
169         }
170         Style.Language = OldLanguage;
171       }
172     }
173 
174     IO.mapOptional("AccessModifierOffset", Style.AccessModifierOffset);
175     IO.mapOptional("AlignAfterOpenBracket", Style.AlignAfterOpenBracket);
176     IO.mapOptional("AlignEscapedNewlinesLeft", Style.AlignEscapedNewlinesLeft);
177     IO.mapOptional("AlignOperands", Style.AlignOperands);
178     IO.mapOptional("AlignTrailingComments", Style.AlignTrailingComments);
179     IO.mapOptional("AllowAllParametersOfDeclarationOnNextLine",
180                    Style.AllowAllParametersOfDeclarationOnNextLine);
181     IO.mapOptional("AllowShortBlocksOnASingleLine",
182                    Style.AllowShortBlocksOnASingleLine);
183     IO.mapOptional("AllowShortCaseLabelsOnASingleLine",
184                    Style.AllowShortCaseLabelsOnASingleLine);
185     IO.mapOptional("AllowShortIfStatementsOnASingleLine",
186                    Style.AllowShortIfStatementsOnASingleLine);
187     IO.mapOptional("AllowShortLoopsOnASingleLine",
188                    Style.AllowShortLoopsOnASingleLine);
189     IO.mapOptional("AllowShortFunctionsOnASingleLine",
190                    Style.AllowShortFunctionsOnASingleLine);
191     IO.mapOptional("AlwaysBreakAfterDefinitionReturnType",
192                    Style.AlwaysBreakAfterDefinitionReturnType);
193     IO.mapOptional("AlwaysBreakTemplateDeclarations",
194                    Style.AlwaysBreakTemplateDeclarations);
195     IO.mapOptional("AlwaysBreakBeforeMultilineStrings",
196                    Style.AlwaysBreakBeforeMultilineStrings);
197     IO.mapOptional("BreakBeforeBinaryOperators",
198                    Style.BreakBeforeBinaryOperators);
199     IO.mapOptional("BreakBeforeTernaryOperators",
200                    Style.BreakBeforeTernaryOperators);
201     IO.mapOptional("BreakConstructorInitializersBeforeComma",
202                    Style.BreakConstructorInitializersBeforeComma);
203     IO.mapOptional("BinPackParameters", Style.BinPackParameters);
204     IO.mapOptional("BinPackArguments", Style.BinPackArguments);
205     IO.mapOptional("ColumnLimit", Style.ColumnLimit);
206     IO.mapOptional("ConstructorInitializerAllOnOneLineOrOnePerLine",
207                    Style.ConstructorInitializerAllOnOneLineOrOnePerLine);
208     IO.mapOptional("ConstructorInitializerIndentWidth",
209                    Style.ConstructorInitializerIndentWidth);
210     IO.mapOptional("DerivePointerAlignment", Style.DerivePointerAlignment);
211     IO.mapOptional("ExperimentalAutoDetectBinPacking",
212                    Style.ExperimentalAutoDetectBinPacking);
213     IO.mapOptional("IndentCaseLabels", Style.IndentCaseLabels);
214     IO.mapOptional("IndentWrappedFunctionNames",
215                    Style.IndentWrappedFunctionNames);
216     IO.mapOptional("IndentFunctionDeclarationAfterType",
217                    Style.IndentWrappedFunctionNames);
218     IO.mapOptional("MaxEmptyLinesToKeep", Style.MaxEmptyLinesToKeep);
219     IO.mapOptional("KeepEmptyLinesAtTheStartOfBlocks",
220                    Style.KeepEmptyLinesAtTheStartOfBlocks);
221     IO.mapOptional("NamespaceIndentation", Style.NamespaceIndentation);
222     IO.mapOptional("ObjCBlockIndentWidth", Style.ObjCBlockIndentWidth);
223     IO.mapOptional("ObjCSpaceAfterProperty", Style.ObjCSpaceAfterProperty);
224     IO.mapOptional("ObjCSpaceBeforeProtocolList",
225                    Style.ObjCSpaceBeforeProtocolList);
226     IO.mapOptional("PenaltyBreakBeforeFirstCallParameter",
227                    Style.PenaltyBreakBeforeFirstCallParameter);
228     IO.mapOptional("PenaltyBreakComment", Style.PenaltyBreakComment);
229     IO.mapOptional("PenaltyBreakString", Style.PenaltyBreakString);
230     IO.mapOptional("PenaltyBreakFirstLessLess",
231                    Style.PenaltyBreakFirstLessLess);
232     IO.mapOptional("PenaltyExcessCharacter", Style.PenaltyExcessCharacter);
233     IO.mapOptional("PenaltyReturnTypeOnItsOwnLine",
234                    Style.PenaltyReturnTypeOnItsOwnLine);
235     IO.mapOptional("PointerAlignment", Style.PointerAlignment);
236     IO.mapOptional("SpacesBeforeTrailingComments",
237                    Style.SpacesBeforeTrailingComments);
238     IO.mapOptional("Cpp11BracedListStyle", Style.Cpp11BracedListStyle);
239     IO.mapOptional("Standard", Style.Standard);
240     IO.mapOptional("IndentWidth", Style.IndentWidth);
241     IO.mapOptional("TabWidth", Style.TabWidth);
242     IO.mapOptional("UseTab", Style.UseTab);
243     IO.mapOptional("BreakBeforeBraces", Style.BreakBeforeBraces);
244     IO.mapOptional("SpacesInParentheses", Style.SpacesInParentheses);
245     IO.mapOptional("SpacesInSquareBrackets", Style.SpacesInSquareBrackets);
246     IO.mapOptional("SpacesInAngles", Style.SpacesInAngles);
247     IO.mapOptional("SpaceInEmptyParentheses", Style.SpaceInEmptyParentheses);
248     IO.mapOptional("SpacesInCStyleCastParentheses",
249                    Style.SpacesInCStyleCastParentheses);
250     IO.mapOptional("SpaceAfterCStyleCast", Style.SpaceAfterCStyleCast);
251     IO.mapOptional("SpacesInContainerLiterals",
252                    Style.SpacesInContainerLiterals);
253     IO.mapOptional("SpaceBeforeAssignmentOperators",
254                    Style.SpaceBeforeAssignmentOperators);
255     IO.mapOptional("ContinuationIndentWidth", Style.ContinuationIndentWidth);
256     IO.mapOptional("CommentPragmas", Style.CommentPragmas);
257     IO.mapOptional("ForEachMacros", Style.ForEachMacros);
258 
259     // For backward compatibility.
260     if (!IO.outputting()) {
261       IO.mapOptional("SpaceAfterControlStatementKeyword",
262                      Style.SpaceBeforeParens);
263       IO.mapOptional("PointerBindsToType", Style.PointerAlignment);
264       IO.mapOptional("DerivePointerBinding", Style.DerivePointerAlignment);
265     }
266     IO.mapOptional("SpaceBeforeParens", Style.SpaceBeforeParens);
267     IO.mapOptional("DisableFormat", Style.DisableFormat);
268   }
269 };
270 
271 // Allows to read vector<FormatStyle> while keeping default values.
272 // IO.getContext() should contain a pointer to the FormatStyle structure, that
273 // will be used to get default values for missing keys.
274 // If the first element has no Language specified, it will be treated as the
275 // default one for the following elements.
276 template <> struct DocumentListTraits<std::vector<FormatStyle> > {
277   static size_t size(IO &IO, std::vector<FormatStyle> &Seq) {
278     return Seq.size();
279   }
280   static FormatStyle &element(IO &IO, std::vector<FormatStyle> &Seq,
281                               size_t Index) {
282     if (Index >= Seq.size()) {
283       assert(Index == Seq.size());
284       FormatStyle Template;
285       if (Seq.size() > 0 && Seq[0].Language == FormatStyle::LK_None) {
286         Template = Seq[0];
287       } else {
288         Template = *((const FormatStyle *)IO.getContext());
289         Template.Language = FormatStyle::LK_None;
290       }
291       Seq.resize(Index + 1, Template);
292     }
293     return Seq[Index];
294   }
295 };
296 }
297 }
298 
299 namespace clang {
300 namespace format {
301 
302 const std::error_category &getParseCategory() {
303   static ParseErrorCategory C;
304   return C;
305 }
306 std::error_code make_error_code(ParseError e) {
307   return std::error_code(static_cast<int>(e), getParseCategory());
308 }
309 
310 const char *ParseErrorCategory::name() const LLVM_NOEXCEPT {
311   return "clang-format.parse_error";
312 }
313 
314 std::string ParseErrorCategory::message(int EV) const {
315   switch (static_cast<ParseError>(EV)) {
316   case ParseError::Success:
317     return "Success";
318   case ParseError::Error:
319     return "Invalid argument";
320   case ParseError::Unsuitable:
321     return "Unsuitable";
322   }
323   llvm_unreachable("unexpected parse error");
324 }
325 
326 FormatStyle getLLVMStyle() {
327   FormatStyle LLVMStyle;
328   LLVMStyle.Language = FormatStyle::LK_Cpp;
329   LLVMStyle.AccessModifierOffset = -2;
330   LLVMStyle.AlignEscapedNewlinesLeft = false;
331   LLVMStyle.AlignAfterOpenBracket = true;
332   LLVMStyle.AlignOperands = true;
333   LLVMStyle.AlignTrailingComments = true;
334   LLVMStyle.AllowAllParametersOfDeclarationOnNextLine = true;
335   LLVMStyle.AllowShortFunctionsOnASingleLine = FormatStyle::SFS_All;
336   LLVMStyle.AllowShortBlocksOnASingleLine = false;
337   LLVMStyle.AllowShortCaseLabelsOnASingleLine = false;
338   LLVMStyle.AllowShortIfStatementsOnASingleLine = false;
339   LLVMStyle.AllowShortLoopsOnASingleLine = false;
340   LLVMStyle.AlwaysBreakAfterDefinitionReturnType = false;
341   LLVMStyle.AlwaysBreakBeforeMultilineStrings = false;
342   LLVMStyle.AlwaysBreakTemplateDeclarations = false;
343   LLVMStyle.BinPackParameters = true;
344   LLVMStyle.BinPackArguments = true;
345   LLVMStyle.BreakBeforeBinaryOperators = FormatStyle::BOS_None;
346   LLVMStyle.BreakBeforeTernaryOperators = true;
347   LLVMStyle.BreakBeforeBraces = FormatStyle::BS_Attach;
348   LLVMStyle.BreakConstructorInitializersBeforeComma = false;
349   LLVMStyle.ColumnLimit = 80;
350   LLVMStyle.CommentPragmas = "^ IWYU pragma:";
351   LLVMStyle.ConstructorInitializerAllOnOneLineOrOnePerLine = false;
352   LLVMStyle.ConstructorInitializerIndentWidth = 4;
353   LLVMStyle.ContinuationIndentWidth = 4;
354   LLVMStyle.Cpp11BracedListStyle = true;
355   LLVMStyle.DerivePointerAlignment = false;
356   LLVMStyle.ExperimentalAutoDetectBinPacking = false;
357   LLVMStyle.ForEachMacros.push_back("foreach");
358   LLVMStyle.ForEachMacros.push_back("Q_FOREACH");
359   LLVMStyle.ForEachMacros.push_back("BOOST_FOREACH");
360   LLVMStyle.IndentCaseLabels = false;
361   LLVMStyle.IndentWrappedFunctionNames = false;
362   LLVMStyle.IndentWidth = 2;
363   LLVMStyle.TabWidth = 8;
364   LLVMStyle.MaxEmptyLinesToKeep = 1;
365   LLVMStyle.KeepEmptyLinesAtTheStartOfBlocks = true;
366   LLVMStyle.NamespaceIndentation = FormatStyle::NI_None;
367   LLVMStyle.ObjCBlockIndentWidth = 2;
368   LLVMStyle.ObjCSpaceAfterProperty = false;
369   LLVMStyle.ObjCSpaceBeforeProtocolList = true;
370   LLVMStyle.PointerAlignment = FormatStyle::PAS_Right;
371   LLVMStyle.SpacesBeforeTrailingComments = 1;
372   LLVMStyle.Standard = FormatStyle::LS_Cpp11;
373   LLVMStyle.UseTab = FormatStyle::UT_Never;
374   LLVMStyle.SpacesInParentheses = false;
375   LLVMStyle.SpacesInSquareBrackets = false;
376   LLVMStyle.SpaceInEmptyParentheses = false;
377   LLVMStyle.SpacesInContainerLiterals = true;
378   LLVMStyle.SpacesInCStyleCastParentheses = false;
379   LLVMStyle.SpaceAfterCStyleCast = false;
380   LLVMStyle.SpaceBeforeParens = FormatStyle::SBPO_ControlStatements;
381   LLVMStyle.SpaceBeforeAssignmentOperators = true;
382   LLVMStyle.SpacesInAngles = false;
383 
384   LLVMStyle.PenaltyBreakComment = 300;
385   LLVMStyle.PenaltyBreakFirstLessLess = 120;
386   LLVMStyle.PenaltyBreakString = 1000;
387   LLVMStyle.PenaltyExcessCharacter = 1000000;
388   LLVMStyle.PenaltyReturnTypeOnItsOwnLine = 60;
389   LLVMStyle.PenaltyBreakBeforeFirstCallParameter = 19;
390 
391   LLVMStyle.DisableFormat = false;
392 
393   return LLVMStyle;
394 }
395 
396 FormatStyle getGoogleStyle(FormatStyle::LanguageKind Language) {
397   FormatStyle GoogleStyle = getLLVMStyle();
398   GoogleStyle.Language = Language;
399 
400   GoogleStyle.AccessModifierOffset = -1;
401   GoogleStyle.AlignEscapedNewlinesLeft = true;
402   GoogleStyle.AllowShortIfStatementsOnASingleLine = true;
403   GoogleStyle.AllowShortLoopsOnASingleLine = true;
404   GoogleStyle.AlwaysBreakBeforeMultilineStrings = true;
405   GoogleStyle.AlwaysBreakTemplateDeclarations = true;
406   GoogleStyle.ConstructorInitializerAllOnOneLineOrOnePerLine = true;
407   GoogleStyle.DerivePointerAlignment = true;
408   GoogleStyle.IndentCaseLabels = true;
409   GoogleStyle.KeepEmptyLinesAtTheStartOfBlocks = false;
410   GoogleStyle.ObjCSpaceAfterProperty = false;
411   GoogleStyle.ObjCSpaceBeforeProtocolList = false;
412   GoogleStyle.PointerAlignment = FormatStyle::PAS_Left;
413   GoogleStyle.SpacesBeforeTrailingComments = 2;
414   GoogleStyle.Standard = FormatStyle::LS_Auto;
415 
416   GoogleStyle.PenaltyReturnTypeOnItsOwnLine = 200;
417   GoogleStyle.PenaltyBreakBeforeFirstCallParameter = 1;
418 
419   if (Language == FormatStyle::LK_Java) {
420     GoogleStyle.AlignAfterOpenBracket = false;
421     GoogleStyle.AlignOperands = false;
422     GoogleStyle.AllowShortFunctionsOnASingleLine = FormatStyle::SFS_Empty;
423     GoogleStyle.BreakBeforeBinaryOperators = FormatStyle::BOS_NonAssignment;
424     GoogleStyle.ColumnLimit = 100;
425     GoogleStyle.SpaceAfterCStyleCast = true;
426     GoogleStyle.SpacesBeforeTrailingComments = 1;
427   } else if (Language == FormatStyle::LK_JavaScript) {
428     GoogleStyle.BreakBeforeTernaryOperators = false;
429     GoogleStyle.MaxEmptyLinesToKeep = 3;
430     GoogleStyle.SpacesInContainerLiterals = false;
431     GoogleStyle.AllowShortFunctionsOnASingleLine = FormatStyle::SFS_Inline;
432   } else if (Language == FormatStyle::LK_Proto) {
433     GoogleStyle.AllowShortFunctionsOnASingleLine = FormatStyle::SFS_None;
434     GoogleStyle.SpacesInContainerLiterals = false;
435   }
436 
437   return GoogleStyle;
438 }
439 
440 FormatStyle getChromiumStyle(FormatStyle::LanguageKind Language) {
441   FormatStyle ChromiumStyle = getGoogleStyle(Language);
442   if (Language == FormatStyle::LK_Java) {
443     ChromiumStyle.IndentWidth = 4;
444     ChromiumStyle.ContinuationIndentWidth = 8;
445   } else {
446     ChromiumStyle.AllowAllParametersOfDeclarationOnNextLine = false;
447     ChromiumStyle.AllowShortFunctionsOnASingleLine = FormatStyle::SFS_Inline;
448     ChromiumStyle.AllowShortIfStatementsOnASingleLine = false;
449     ChromiumStyle.AllowShortLoopsOnASingleLine = false;
450     ChromiumStyle.BinPackParameters = false;
451     ChromiumStyle.DerivePointerAlignment = false;
452   }
453   return ChromiumStyle;
454 }
455 
456 FormatStyle getMozillaStyle() {
457   FormatStyle MozillaStyle = getLLVMStyle();
458   MozillaStyle.AllowAllParametersOfDeclarationOnNextLine = false;
459   MozillaStyle.Cpp11BracedListStyle = false;
460   MozillaStyle.ConstructorInitializerAllOnOneLineOrOnePerLine = true;
461   MozillaStyle.DerivePointerAlignment = true;
462   MozillaStyle.IndentCaseLabels = true;
463   MozillaStyle.ObjCSpaceAfterProperty = true;
464   MozillaStyle.ObjCSpaceBeforeProtocolList = false;
465   MozillaStyle.PenaltyReturnTypeOnItsOwnLine = 200;
466   MozillaStyle.PointerAlignment = FormatStyle::PAS_Left;
467   MozillaStyle.Standard = FormatStyle::LS_Cpp03;
468   return MozillaStyle;
469 }
470 
471 FormatStyle getWebKitStyle() {
472   FormatStyle Style = getLLVMStyle();
473   Style.AccessModifierOffset = -4;
474   Style.AlignAfterOpenBracket = false;
475   Style.AlignOperands = false;
476   Style.AlignTrailingComments = false;
477   Style.BreakBeforeBinaryOperators = FormatStyle::BOS_All;
478   Style.BreakBeforeBraces = FormatStyle::BS_Stroustrup;
479   Style.BreakConstructorInitializersBeforeComma = true;
480   Style.Cpp11BracedListStyle = false;
481   Style.ColumnLimit = 0;
482   Style.IndentWidth = 4;
483   Style.NamespaceIndentation = FormatStyle::NI_Inner;
484   Style.ObjCBlockIndentWidth = 4;
485   Style.ObjCSpaceAfterProperty = true;
486   Style.PointerAlignment = FormatStyle::PAS_Left;
487   Style.Standard = FormatStyle::LS_Cpp03;
488   return Style;
489 }
490 
491 FormatStyle getGNUStyle() {
492   FormatStyle Style = getLLVMStyle();
493   Style.AlwaysBreakAfterDefinitionReturnType = true;
494   Style.BreakBeforeBinaryOperators = FormatStyle::BOS_All;
495   Style.BreakBeforeBraces = FormatStyle::BS_GNU;
496   Style.BreakBeforeTernaryOperators = true;
497   Style.Cpp11BracedListStyle = false;
498   Style.ColumnLimit = 79;
499   Style.SpaceBeforeParens = FormatStyle::SBPO_Always;
500   Style.Standard = FormatStyle::LS_Cpp03;
501   return Style;
502 }
503 
504 FormatStyle getNoStyle() {
505   FormatStyle NoStyle = getLLVMStyle();
506   NoStyle.DisableFormat = true;
507   return NoStyle;
508 }
509 
510 bool getPredefinedStyle(StringRef Name, FormatStyle::LanguageKind Language,
511                         FormatStyle *Style) {
512   if (Name.equals_lower("llvm")) {
513     *Style = getLLVMStyle();
514   } else if (Name.equals_lower("chromium")) {
515     *Style = getChromiumStyle(Language);
516   } else if (Name.equals_lower("mozilla")) {
517     *Style = getMozillaStyle();
518   } else if (Name.equals_lower("google")) {
519     *Style = getGoogleStyle(Language);
520   } else if (Name.equals_lower("webkit")) {
521     *Style = getWebKitStyle();
522   } else if (Name.equals_lower("gnu")) {
523     *Style = getGNUStyle();
524   } else if (Name.equals_lower("none")) {
525     *Style = getNoStyle();
526   } else {
527     return false;
528   }
529 
530   Style->Language = Language;
531   return true;
532 }
533 
534 std::error_code parseConfiguration(StringRef Text, FormatStyle *Style) {
535   assert(Style);
536   FormatStyle::LanguageKind Language = Style->Language;
537   assert(Language != FormatStyle::LK_None);
538   if (Text.trim().empty())
539     return make_error_code(ParseError::Error);
540 
541   std::vector<FormatStyle> Styles;
542   llvm::yaml::Input Input(Text);
543   // DocumentListTraits<vector<FormatStyle>> uses the context to get default
544   // values for the fields, keys for which are missing from the configuration.
545   // Mapping also uses the context to get the language to find the correct
546   // base style.
547   Input.setContext(Style);
548   Input >> Styles;
549   if (Input.error())
550     return Input.error();
551 
552   for (unsigned i = 0; i < Styles.size(); ++i) {
553     // Ensures that only the first configuration can skip the Language option.
554     if (Styles[i].Language == FormatStyle::LK_None && i != 0)
555       return make_error_code(ParseError::Error);
556     // Ensure that each language is configured at most once.
557     for (unsigned j = 0; j < i; ++j) {
558       if (Styles[i].Language == Styles[j].Language) {
559         DEBUG(llvm::dbgs()
560               << "Duplicate languages in the config file on positions " << j
561               << " and " << i << "\n");
562         return make_error_code(ParseError::Error);
563       }
564     }
565   }
566   // Look for a suitable configuration starting from the end, so we can
567   // find the configuration for the specific language first, and the default
568   // configuration (which can only be at slot 0) after it.
569   for (int i = Styles.size() - 1; i >= 0; --i) {
570     if (Styles[i].Language == Language ||
571         Styles[i].Language == FormatStyle::LK_None) {
572       *Style = Styles[i];
573       Style->Language = Language;
574       return make_error_code(ParseError::Success);
575     }
576   }
577   return make_error_code(ParseError::Unsuitable);
578 }
579 
580 std::string configurationAsText(const FormatStyle &Style) {
581   std::string Text;
582   llvm::raw_string_ostream Stream(Text);
583   llvm::yaml::Output Output(Stream);
584   // We use the same mapping method for input and output, so we need a non-const
585   // reference here.
586   FormatStyle NonConstStyle = Style;
587   Output << NonConstStyle;
588   return Stream.str();
589 }
590 
591 namespace {
592 
593 class FormatTokenLexer {
594 public:
595   FormatTokenLexer(SourceManager &SourceMgr, FileID ID, FormatStyle &Style,
596                    encoding::Encoding Encoding)
597       : FormatTok(nullptr), IsFirstToken(true), GreaterStashed(false),
598         Column(0), TrailingWhitespace(0), SourceMgr(SourceMgr), ID(ID),
599         Style(Style), IdentTable(getFormattingLangOpts(Style)),
600         Keywords(IdentTable), Encoding(Encoding), FirstInLineIndex(0),
601         FormattingDisabled(false) {
602     Lex.reset(new Lexer(ID, SourceMgr.getBuffer(ID), SourceMgr,
603                         getFormattingLangOpts(Style)));
604     Lex->SetKeepWhitespaceMode(true);
605 
606     for (const std::string &ForEachMacro : Style.ForEachMacros)
607       ForEachMacros.push_back(&IdentTable.get(ForEachMacro));
608     std::sort(ForEachMacros.begin(), ForEachMacros.end());
609   }
610 
611   ArrayRef<FormatToken *> lex() {
612     assert(Tokens.empty());
613     assert(FirstInLineIndex == 0);
614     do {
615       Tokens.push_back(getNextToken());
616       tryMergePreviousTokens();
617       if (Tokens.back()->NewlinesBefore > 0)
618         FirstInLineIndex = Tokens.size() - 1;
619     } while (Tokens.back()->Tok.isNot(tok::eof));
620     return Tokens;
621   }
622 
623   const AdditionalKeywords &getKeywords() { return Keywords; }
624 
625 private:
626   void tryMergePreviousTokens() {
627     if (tryMerge_TMacro())
628       return;
629     if (tryMergeConflictMarkers())
630       return;
631 
632     if (Style.Language == FormatStyle::LK_JavaScript) {
633       if (tryMergeJSRegexLiteral())
634         return;
635       if (tryMergeEscapeSequence())
636         return;
637 
638       static tok::TokenKind JSIdentity[] = { tok::equalequal, tok::equal };
639       static tok::TokenKind JSNotIdentity[] = { tok::exclaimequal, tok::equal };
640       static tok::TokenKind JSShiftEqual[] = { tok::greater, tok::greater,
641                                                tok::greaterequal };
642       static tok::TokenKind JSRightArrow[] = { tok::equal, tok::greater };
643       // FIXME: We probably need to change token type to mimic operator with the
644       // correct priority.
645       if (tryMergeTokens(JSIdentity))
646         return;
647       if (tryMergeTokens(JSNotIdentity))
648         return;
649       if (tryMergeTokens(JSShiftEqual))
650         return;
651       if (tryMergeTokens(JSRightArrow))
652         return;
653     }
654   }
655 
656   bool tryMergeTokens(ArrayRef<tok::TokenKind> Kinds) {
657     if (Tokens.size() < Kinds.size())
658       return false;
659 
660     SmallVectorImpl<FormatToken *>::const_iterator First =
661         Tokens.end() - Kinds.size();
662     if (!First[0]->is(Kinds[0]))
663       return false;
664     unsigned AddLength = 0;
665     for (unsigned i = 1; i < Kinds.size(); ++i) {
666       if (!First[i]->is(Kinds[i]) || First[i]->WhitespaceRange.getBegin() !=
667                                          First[i]->WhitespaceRange.getEnd())
668         return false;
669       AddLength += First[i]->TokenText.size();
670     }
671     Tokens.resize(Tokens.size() - Kinds.size() + 1);
672     First[0]->TokenText = StringRef(First[0]->TokenText.data(),
673                                     First[0]->TokenText.size() + AddLength);
674     First[0]->ColumnWidth += AddLength;
675     return true;
676   }
677 
678   // Tries to merge an escape sequence, i.e. a "\\" and the following
679   // character. Use e.g. inside JavaScript regex literals.
680   bool tryMergeEscapeSequence() {
681     if (Tokens.size() < 2)
682       return false;
683     FormatToken *Previous = Tokens[Tokens.size() - 2];
684     if (Previous->isNot(tok::unknown) || Previous->TokenText != "\\")
685       return false;
686     ++Previous->ColumnWidth;
687     StringRef Text = Previous->TokenText;
688     Previous->TokenText = StringRef(Text.data(), Text.size() + 1);
689     resetLexer(SourceMgr.getFileOffset(Tokens.back()->Tok.getLocation()) + 1);
690     Tokens.resize(Tokens.size() - 1);
691     Column = Previous->OriginalColumn + Previous->ColumnWidth;
692     return true;
693   }
694 
695   // Try to determine whether the current token ends a JavaScript regex literal.
696   // We heuristically assume that this is a regex literal if we find two
697   // unescaped slashes on a line and the token before the first slash is one of
698   // "(;,{}![:?", a binary operator or 'return', as those cannot be followed by
699   // a division.
700   bool tryMergeJSRegexLiteral() {
701     if (Tokens.size() < 2)
702       return false;
703     // If a regex literal ends in "\//", this gets represented by an unknown
704     // token "\" and a comment.
705     bool MightEndWithEscapedSlash =
706         Tokens.back()->is(tok::comment) &&
707         Tokens.back()->TokenText.startswith("//") &&
708         Tokens[Tokens.size() - 2]->TokenText == "\\";
709     if (!MightEndWithEscapedSlash &&
710         (Tokens.back()->isNot(tok::slash) ||
711          (Tokens[Tokens.size() - 2]->is(tok::unknown) &&
712           Tokens[Tokens.size() - 2]->TokenText == "\\")))
713       return false;
714     unsigned TokenCount = 0;
715     unsigned LastColumn = Tokens.back()->OriginalColumn;
716     for (auto I = Tokens.rbegin() + 1, E = Tokens.rend(); I != E; ++I) {
717       ++TokenCount;
718       if (I[0]->is(tok::slash) && I + 1 != E &&
719           (I[1]->isOneOf(tok::l_paren, tok::semi, tok::l_brace, tok::r_brace,
720                          tok::exclaim, tok::l_square, tok::colon, tok::comma,
721                          tok::question, tok::kw_return) ||
722            I[1]->isBinaryOperator())) {
723         if (MightEndWithEscapedSlash) {
724           // This regex literal ends in '\//'. Skip past the '//' of the last
725           // token and re-start lexing from there.
726           SourceLocation Loc = Tokens.back()->Tok.getLocation();
727           resetLexer(SourceMgr.getFileOffset(Loc) + 2);
728         }
729         Tokens.resize(Tokens.size() - TokenCount);
730         Tokens.back()->Tok.setKind(tok::unknown);
731         Tokens.back()->Type = TT_RegexLiteral;
732         Tokens.back()->ColumnWidth += LastColumn - I[0]->OriginalColumn;
733         return true;
734       }
735 
736       // There can't be a newline inside a regex literal.
737       if (I[0]->NewlinesBefore > 0)
738         return false;
739     }
740     return false;
741   }
742 
743   bool tryMerge_TMacro() {
744     if (Tokens.size() < 4)
745       return false;
746     FormatToken *Last = Tokens.back();
747     if (!Last->is(tok::r_paren))
748       return false;
749 
750     FormatToken *String = Tokens[Tokens.size() - 2];
751     if (!String->is(tok::string_literal) || String->IsMultiline)
752       return false;
753 
754     if (!Tokens[Tokens.size() - 3]->is(tok::l_paren))
755       return false;
756 
757     FormatToken *Macro = Tokens[Tokens.size() - 4];
758     if (Macro->TokenText != "_T")
759       return false;
760 
761     const char *Start = Macro->TokenText.data();
762     const char *End = Last->TokenText.data() + Last->TokenText.size();
763     String->TokenText = StringRef(Start, End - Start);
764     String->IsFirst = Macro->IsFirst;
765     String->LastNewlineOffset = Macro->LastNewlineOffset;
766     String->WhitespaceRange = Macro->WhitespaceRange;
767     String->OriginalColumn = Macro->OriginalColumn;
768     String->ColumnWidth = encoding::columnWidthWithTabs(
769         String->TokenText, String->OriginalColumn, Style.TabWidth, Encoding);
770 
771     Tokens.pop_back();
772     Tokens.pop_back();
773     Tokens.pop_back();
774     Tokens.back() = String;
775     return true;
776   }
777 
778   bool tryMergeConflictMarkers() {
779     if (Tokens.back()->NewlinesBefore == 0 && Tokens.back()->isNot(tok::eof))
780       return false;
781 
782     // Conflict lines look like:
783     // <marker> <text from the vcs>
784     // For example:
785     // >>>>>>> /file/in/file/system at revision 1234
786     //
787     // We merge all tokens in a line that starts with a conflict marker
788     // into a single token with a special token type that the unwrapped line
789     // parser will use to correctly rebuild the underlying code.
790 
791     FileID ID;
792     // Get the position of the first token in the line.
793     unsigned FirstInLineOffset;
794     std::tie(ID, FirstInLineOffset) = SourceMgr.getDecomposedLoc(
795         Tokens[FirstInLineIndex]->getStartOfNonWhitespace());
796     StringRef Buffer = SourceMgr.getBuffer(ID)->getBuffer();
797     // Calculate the offset of the start of the current line.
798     auto LineOffset = Buffer.rfind('\n', FirstInLineOffset);
799     if (LineOffset == StringRef::npos) {
800       LineOffset = 0;
801     } else {
802       ++LineOffset;
803     }
804 
805     auto FirstSpace = Buffer.find_first_of(" \n", LineOffset);
806     StringRef LineStart;
807     if (FirstSpace == StringRef::npos) {
808       LineStart = Buffer.substr(LineOffset);
809     } else {
810       LineStart = Buffer.substr(LineOffset, FirstSpace - LineOffset);
811     }
812 
813     TokenType Type = TT_Unknown;
814     if (LineStart == "<<<<<<<" || LineStart == ">>>>") {
815       Type = TT_ConflictStart;
816     } else if (LineStart == "|||||||" || LineStart == "=======" ||
817                LineStart == "====") {
818       Type = TT_ConflictAlternative;
819     } else if (LineStart == ">>>>>>>" || LineStart == "<<<<") {
820       Type = TT_ConflictEnd;
821     }
822 
823     if (Type != TT_Unknown) {
824       FormatToken *Next = Tokens.back();
825 
826       Tokens.resize(FirstInLineIndex + 1);
827       // We do not need to build a complete token here, as we will skip it
828       // during parsing anyway (as we must not touch whitespace around conflict
829       // markers).
830       Tokens.back()->Type = Type;
831       Tokens.back()->Tok.setKind(tok::kw___unknown_anytype);
832 
833       Tokens.push_back(Next);
834       return true;
835     }
836 
837     return false;
838   }
839 
840   FormatToken *getNextToken() {
841     if (GreaterStashed) {
842       // Create a synthesized second '>' token.
843       // FIXME: Increment Column and set OriginalColumn.
844       Token Greater = FormatTok->Tok;
845       FormatTok = new (Allocator.Allocate()) FormatToken;
846       FormatTok->Tok = Greater;
847       SourceLocation GreaterLocation =
848           FormatTok->Tok.getLocation().getLocWithOffset(1);
849       FormatTok->WhitespaceRange =
850           SourceRange(GreaterLocation, GreaterLocation);
851       FormatTok->TokenText = ">";
852       FormatTok->ColumnWidth = 1;
853       GreaterStashed = false;
854       return FormatTok;
855     }
856 
857     FormatTok = new (Allocator.Allocate()) FormatToken;
858     readRawToken(*FormatTok);
859     SourceLocation WhitespaceStart =
860         FormatTok->Tok.getLocation().getLocWithOffset(-TrailingWhitespace);
861     FormatTok->IsFirst = IsFirstToken;
862     IsFirstToken = false;
863 
864     // Consume and record whitespace until we find a significant token.
865     unsigned WhitespaceLength = TrailingWhitespace;
866     while (FormatTok->Tok.is(tok::unknown)) {
867       for (int i = 0, e = FormatTok->TokenText.size(); i != e; ++i) {
868         switch (FormatTok->TokenText[i]) {
869         case '\n':
870           ++FormatTok->NewlinesBefore;
871           // FIXME: This is technically incorrect, as it could also
872           // be a literal backslash at the end of the line.
873           if (i == 0 || (FormatTok->TokenText[i - 1] != '\\' &&
874                          (FormatTok->TokenText[i - 1] != '\r' || i == 1 ||
875                           FormatTok->TokenText[i - 2] != '\\')))
876             FormatTok->HasUnescapedNewline = true;
877           FormatTok->LastNewlineOffset = WhitespaceLength + i + 1;
878           Column = 0;
879           break;
880         case '\r':
881         case '\f':
882         case '\v':
883           Column = 0;
884           break;
885         case ' ':
886           ++Column;
887           break;
888         case '\t':
889           Column += Style.TabWidth - Column % Style.TabWidth;
890           break;
891         case '\\':
892           if (i + 1 == e || (FormatTok->TokenText[i + 1] != '\r' &&
893                              FormatTok->TokenText[i + 1] != '\n'))
894             FormatTok->Type = TT_ImplicitStringLiteral;
895           break;
896         default:
897           FormatTok->Type = TT_ImplicitStringLiteral;
898           ++Column;
899           break;
900         }
901       }
902 
903       if (FormatTok->is(TT_ImplicitStringLiteral))
904         break;
905       WhitespaceLength += FormatTok->Tok.getLength();
906 
907       readRawToken(*FormatTok);
908     }
909 
910     // In case the token starts with escaped newlines, we want to
911     // take them into account as whitespace - this pattern is quite frequent
912     // in macro definitions.
913     // FIXME: Add a more explicit test.
914     while (FormatTok->TokenText.size() > 1 && FormatTok->TokenText[0] == '\\' &&
915            FormatTok->TokenText[1] == '\n') {
916       ++FormatTok->NewlinesBefore;
917       WhitespaceLength += 2;
918       Column = 0;
919       FormatTok->TokenText = FormatTok->TokenText.substr(2);
920     }
921 
922     FormatTok->WhitespaceRange = SourceRange(
923         WhitespaceStart, WhitespaceStart.getLocWithOffset(WhitespaceLength));
924 
925     FormatTok->OriginalColumn = Column;
926 
927     TrailingWhitespace = 0;
928     if (FormatTok->Tok.is(tok::comment)) {
929       // FIXME: Add the trimmed whitespace to Column.
930       StringRef UntrimmedText = FormatTok->TokenText;
931       FormatTok->TokenText = FormatTok->TokenText.rtrim(" \t\v\f");
932       TrailingWhitespace = UntrimmedText.size() - FormatTok->TokenText.size();
933     } else if (FormatTok->Tok.is(tok::raw_identifier)) {
934       IdentifierInfo &Info = IdentTable.get(FormatTok->TokenText);
935       FormatTok->Tok.setIdentifierInfo(&Info);
936       FormatTok->Tok.setKind(Info.getTokenID());
937       if (Style.Language == FormatStyle::LK_Java &&
938           FormatTok->isOneOf(tok::kw_struct, tok::kw_union, tok::kw_delete)) {
939         FormatTok->Tok.setKind(tok::identifier);
940         FormatTok->Tok.setIdentifierInfo(nullptr);
941       }
942     } else if (FormatTok->Tok.is(tok::greatergreater)) {
943       FormatTok->Tok.setKind(tok::greater);
944       FormatTok->TokenText = FormatTok->TokenText.substr(0, 1);
945       GreaterStashed = true;
946     }
947 
948     // Now FormatTok is the next non-whitespace token.
949 
950     StringRef Text = FormatTok->TokenText;
951     size_t FirstNewlinePos = Text.find('\n');
952     if (FirstNewlinePos == StringRef::npos) {
953       // FIXME: ColumnWidth actually depends on the start column, we need to
954       // take this into account when the token is moved.
955       FormatTok->ColumnWidth =
956           encoding::columnWidthWithTabs(Text, Column, Style.TabWidth, Encoding);
957       Column += FormatTok->ColumnWidth;
958     } else {
959       FormatTok->IsMultiline = true;
960       // FIXME: ColumnWidth actually depends on the start column, we need to
961       // take this into account when the token is moved.
962       FormatTok->ColumnWidth = encoding::columnWidthWithTabs(
963           Text.substr(0, FirstNewlinePos), Column, Style.TabWidth, Encoding);
964 
965       // The last line of the token always starts in column 0.
966       // Thus, the length can be precomputed even in the presence of tabs.
967       FormatTok->LastLineColumnWidth = encoding::columnWidthWithTabs(
968           Text.substr(Text.find_last_of('\n') + 1), 0, Style.TabWidth,
969           Encoding);
970       Column = FormatTok->LastLineColumnWidth;
971     }
972 
973     FormatTok->IsForEachMacro =
974         std::binary_search(ForEachMacros.begin(), ForEachMacros.end(),
975                            FormatTok->Tok.getIdentifierInfo());
976 
977     return FormatTok;
978   }
979 
980   FormatToken *FormatTok;
981   bool IsFirstToken;
982   bool GreaterStashed;
983   unsigned Column;
984   unsigned TrailingWhitespace;
985   std::unique_ptr<Lexer> Lex;
986   SourceManager &SourceMgr;
987   FileID ID;
988   FormatStyle &Style;
989   IdentifierTable IdentTable;
990   AdditionalKeywords Keywords;
991   encoding::Encoding Encoding;
992   llvm::SpecificBumpPtrAllocator<FormatToken> Allocator;
993   // Index (in 'Tokens') of the last token that starts a new line.
994   unsigned FirstInLineIndex;
995   SmallVector<FormatToken *, 16> Tokens;
996   SmallVector<IdentifierInfo *, 8> ForEachMacros;
997 
998   bool FormattingDisabled;
999 
1000   void readRawToken(FormatToken &Tok) {
1001     Lex->LexFromRawLexer(Tok.Tok);
1002     Tok.TokenText = StringRef(SourceMgr.getCharacterData(Tok.Tok.getLocation()),
1003                               Tok.Tok.getLength());
1004     // For formatting, treat unterminated string literals like normal string
1005     // literals.
1006     if (Tok.is(tok::unknown)) {
1007       if (!Tok.TokenText.empty() && Tok.TokenText[0] == '"') {
1008         Tok.Tok.setKind(tok::string_literal);
1009         Tok.IsUnterminatedLiteral = true;
1010       } else if (Style.Language == FormatStyle::LK_JavaScript &&
1011                  Tok.TokenText == "''") {
1012         Tok.Tok.setKind(tok::char_constant);
1013       }
1014     }
1015 
1016     if (Tok.is(tok::comment) && (Tok.TokenText == "// clang-format on" ||
1017                                  Tok.TokenText == "/* clang-format on */")) {
1018       FormattingDisabled = false;
1019     }
1020 
1021     Tok.Finalized = FormattingDisabled;
1022 
1023     if (Tok.is(tok::comment) && (Tok.TokenText == "// clang-format off" ||
1024                                  Tok.TokenText == "/* clang-format off */")) {
1025       FormattingDisabled = true;
1026     }
1027   }
1028 
1029   void resetLexer(unsigned Offset) {
1030     StringRef Buffer = SourceMgr.getBufferData(ID);
1031     Lex.reset(new Lexer(SourceMgr.getLocForStartOfFile(ID),
1032                         getFormattingLangOpts(Style), Buffer.begin(),
1033                         Buffer.begin() + Offset, Buffer.end()));
1034     Lex->SetKeepWhitespaceMode(true);
1035   }
1036 };
1037 
1038 static StringRef getLanguageName(FormatStyle::LanguageKind Language) {
1039   switch (Language) {
1040   case FormatStyle::LK_Cpp:
1041     return "C++";
1042   case FormatStyle::LK_Java:
1043     return "Java";
1044   case FormatStyle::LK_JavaScript:
1045     return "JavaScript";
1046   case FormatStyle::LK_Proto:
1047     return "Proto";
1048   default:
1049     return "Unknown";
1050   }
1051 }
1052 
1053 class Formatter : public UnwrappedLineConsumer {
1054 public:
1055   Formatter(const FormatStyle &Style, SourceManager &SourceMgr, FileID ID,
1056             ArrayRef<CharSourceRange> Ranges)
1057       : Style(Style), ID(ID), SourceMgr(SourceMgr),
1058         Whitespaces(SourceMgr, Style,
1059                     inputUsesCRLF(SourceMgr.getBufferData(ID))),
1060         Ranges(Ranges.begin(), Ranges.end()), UnwrappedLines(1),
1061         Encoding(encoding::detectEncoding(SourceMgr.getBufferData(ID))) {
1062     DEBUG(llvm::dbgs() << "File encoding: "
1063                        << (Encoding == encoding::Encoding_UTF8 ? "UTF8"
1064                                                                : "unknown")
1065                        << "\n");
1066     DEBUG(llvm::dbgs() << "Language: " << getLanguageName(Style.Language)
1067                        << "\n");
1068   }
1069 
1070   tooling::Replacements format() {
1071     tooling::Replacements Result;
1072     FormatTokenLexer Tokens(SourceMgr, ID, Style, Encoding);
1073 
1074     UnwrappedLineParser Parser(Style, Tokens.getKeywords(), Tokens.lex(),
1075                                *this);
1076     bool StructuralError = Parser.parse();
1077     assert(UnwrappedLines.rbegin()->empty());
1078     for (unsigned Run = 0, RunE = UnwrappedLines.size(); Run + 1 != RunE;
1079          ++Run) {
1080       DEBUG(llvm::dbgs() << "Run " << Run << "...\n");
1081       SmallVector<AnnotatedLine *, 16> AnnotatedLines;
1082       for (unsigned i = 0, e = UnwrappedLines[Run].size(); i != e; ++i) {
1083         AnnotatedLines.push_back(new AnnotatedLine(UnwrappedLines[Run][i]));
1084       }
1085       tooling::Replacements RunResult =
1086           format(AnnotatedLines, StructuralError, Tokens);
1087       DEBUG({
1088         llvm::dbgs() << "Replacements for run " << Run << ":\n";
1089         for (tooling::Replacements::iterator I = RunResult.begin(),
1090                                              E = RunResult.end();
1091              I != E; ++I) {
1092           llvm::dbgs() << I->toString() << "\n";
1093         }
1094       });
1095       for (unsigned i = 0, e = AnnotatedLines.size(); i != e; ++i) {
1096         delete AnnotatedLines[i];
1097       }
1098       Result.insert(RunResult.begin(), RunResult.end());
1099       Whitespaces.reset();
1100     }
1101     return Result;
1102   }
1103 
1104   tooling::Replacements format(SmallVectorImpl<AnnotatedLine *> &AnnotatedLines,
1105                                bool StructuralError, FormatTokenLexer &Tokens) {
1106     TokenAnnotator Annotator(Style, Tokens.getKeywords());
1107     for (unsigned i = 0, e = AnnotatedLines.size(); i != e; ++i) {
1108       Annotator.annotate(*AnnotatedLines[i]);
1109     }
1110     deriveLocalStyle(AnnotatedLines);
1111     for (unsigned i = 0, e = AnnotatedLines.size(); i != e; ++i) {
1112       Annotator.calculateFormattingInformation(*AnnotatedLines[i]);
1113     }
1114     computeAffectedLines(AnnotatedLines.begin(), AnnotatedLines.end());
1115 
1116     Annotator.setCommentLineLevels(AnnotatedLines);
1117     ContinuationIndenter Indenter(Style, Tokens.getKeywords(), SourceMgr,
1118                                   Whitespaces, Encoding,
1119                                   BinPackInconclusiveFunctions);
1120     UnwrappedLineFormatter Formatter(&Indenter, &Whitespaces, Style);
1121     Formatter.format(AnnotatedLines, /*DryRun=*/false);
1122     return Whitespaces.generateReplacements();
1123   }
1124 
1125 private:
1126   // Determines which lines are affected by the SourceRanges given as input.
1127   // Returns \c true if at least one line between I and E or one of their
1128   // children is affected.
1129   bool computeAffectedLines(SmallVectorImpl<AnnotatedLine *>::iterator I,
1130                             SmallVectorImpl<AnnotatedLine *>::iterator E) {
1131     bool SomeLineAffected = false;
1132     const AnnotatedLine *PreviousLine = nullptr;
1133     while (I != E) {
1134       AnnotatedLine *Line = *I;
1135       Line->LeadingEmptyLinesAffected = affectsLeadingEmptyLines(*Line->First);
1136 
1137       // If a line is part of a preprocessor directive, it needs to be formatted
1138       // if any token within the directive is affected.
1139       if (Line->InPPDirective) {
1140         FormatToken *Last = Line->Last;
1141         SmallVectorImpl<AnnotatedLine *>::iterator PPEnd = I + 1;
1142         while (PPEnd != E && !(*PPEnd)->First->HasUnescapedNewline) {
1143           Last = (*PPEnd)->Last;
1144           ++PPEnd;
1145         }
1146 
1147         if (affectsTokenRange(*Line->First, *Last,
1148                               /*IncludeLeadingNewlines=*/false)) {
1149           SomeLineAffected = true;
1150           markAllAsAffected(I, PPEnd);
1151         }
1152         I = PPEnd;
1153         continue;
1154       }
1155 
1156       if (nonPPLineAffected(Line, PreviousLine))
1157         SomeLineAffected = true;
1158 
1159       PreviousLine = Line;
1160       ++I;
1161     }
1162     return SomeLineAffected;
1163   }
1164 
1165   // Determines whether 'Line' is affected by the SourceRanges given as input.
1166   // Returns \c true if line or one if its children is affected.
1167   bool nonPPLineAffected(AnnotatedLine *Line,
1168                          const AnnotatedLine *PreviousLine) {
1169     bool SomeLineAffected = false;
1170     Line->ChildrenAffected =
1171         computeAffectedLines(Line->Children.begin(), Line->Children.end());
1172     if (Line->ChildrenAffected)
1173       SomeLineAffected = true;
1174 
1175     // Stores whether one of the line's tokens is directly affected.
1176     bool SomeTokenAffected = false;
1177     // Stores whether we need to look at the leading newlines of the next token
1178     // in order to determine whether it was affected.
1179     bool IncludeLeadingNewlines = false;
1180 
1181     // Stores whether the first child line of any of this line's tokens is
1182     // affected.
1183     bool SomeFirstChildAffected = false;
1184 
1185     for (FormatToken *Tok = Line->First; Tok; Tok = Tok->Next) {
1186       // Determine whether 'Tok' was affected.
1187       if (affectsTokenRange(*Tok, *Tok, IncludeLeadingNewlines))
1188         SomeTokenAffected = true;
1189 
1190       // Determine whether the first child of 'Tok' was affected.
1191       if (!Tok->Children.empty() && Tok->Children.front()->Affected)
1192         SomeFirstChildAffected = true;
1193 
1194       IncludeLeadingNewlines = Tok->Children.empty();
1195     }
1196 
1197     // Was this line moved, i.e. has it previously been on the same line as an
1198     // affected line?
1199     bool LineMoved = PreviousLine && PreviousLine->Affected &&
1200                      Line->First->NewlinesBefore == 0;
1201 
1202     bool IsContinuedComment =
1203         Line->First->is(tok::comment) && Line->First->Next == nullptr &&
1204         Line->First->NewlinesBefore < 2 && PreviousLine &&
1205         PreviousLine->Affected && PreviousLine->Last->is(tok::comment);
1206 
1207     if (SomeTokenAffected || SomeFirstChildAffected || LineMoved ||
1208         IsContinuedComment) {
1209       Line->Affected = true;
1210       SomeLineAffected = true;
1211     }
1212     return SomeLineAffected;
1213   }
1214 
1215   // Marks all lines between I and E as well as all their children as affected.
1216   void markAllAsAffected(SmallVectorImpl<AnnotatedLine *>::iterator I,
1217                          SmallVectorImpl<AnnotatedLine *>::iterator E) {
1218     while (I != E) {
1219       (*I)->Affected = true;
1220       markAllAsAffected((*I)->Children.begin(), (*I)->Children.end());
1221       ++I;
1222     }
1223   }
1224 
1225   // Returns true if the range from 'First' to 'Last' intersects with one of the
1226   // input ranges.
1227   bool affectsTokenRange(const FormatToken &First, const FormatToken &Last,
1228                          bool IncludeLeadingNewlines) {
1229     SourceLocation Start = First.WhitespaceRange.getBegin();
1230     if (!IncludeLeadingNewlines)
1231       Start = Start.getLocWithOffset(First.LastNewlineOffset);
1232     SourceLocation End = Last.getStartOfNonWhitespace();
1233     End = End.getLocWithOffset(Last.TokenText.size());
1234     CharSourceRange Range = CharSourceRange::getCharRange(Start, End);
1235     return affectsCharSourceRange(Range);
1236   }
1237 
1238   // Returns true if one of the input ranges intersect the leading empty lines
1239   // before 'Tok'.
1240   bool affectsLeadingEmptyLines(const FormatToken &Tok) {
1241     CharSourceRange EmptyLineRange = CharSourceRange::getCharRange(
1242         Tok.WhitespaceRange.getBegin(),
1243         Tok.WhitespaceRange.getBegin().getLocWithOffset(Tok.LastNewlineOffset));
1244     return affectsCharSourceRange(EmptyLineRange);
1245   }
1246 
1247   // Returns true if 'Range' intersects with one of the input ranges.
1248   bool affectsCharSourceRange(const CharSourceRange &Range) {
1249     for (SmallVectorImpl<CharSourceRange>::const_iterator I = Ranges.begin(),
1250                                                           E = Ranges.end();
1251          I != E; ++I) {
1252       if (!SourceMgr.isBeforeInTranslationUnit(Range.getEnd(), I->getBegin()) &&
1253           !SourceMgr.isBeforeInTranslationUnit(I->getEnd(), Range.getBegin()))
1254         return true;
1255     }
1256     return false;
1257   }
1258 
1259   static bool inputUsesCRLF(StringRef Text) {
1260     return Text.count('\r') * 2 > Text.count('\n');
1261   }
1262 
1263   void
1264   deriveLocalStyle(const SmallVectorImpl<AnnotatedLine *> &AnnotatedLines) {
1265     unsigned CountBoundToVariable = 0;
1266     unsigned CountBoundToType = 0;
1267     bool HasCpp03IncompatibleFormat = false;
1268     bool HasBinPackedFunction = false;
1269     bool HasOnePerLineFunction = false;
1270     for (unsigned i = 0, e = AnnotatedLines.size(); i != e; ++i) {
1271       if (!AnnotatedLines[i]->First->Next)
1272         continue;
1273       FormatToken *Tok = AnnotatedLines[i]->First->Next;
1274       while (Tok->Next) {
1275         if (Tok->is(TT_PointerOrReference)) {
1276           bool SpacesBefore =
1277               Tok->WhitespaceRange.getBegin() != Tok->WhitespaceRange.getEnd();
1278           bool SpacesAfter = Tok->Next->WhitespaceRange.getBegin() !=
1279                              Tok->Next->WhitespaceRange.getEnd();
1280           if (SpacesBefore && !SpacesAfter)
1281             ++CountBoundToVariable;
1282           else if (!SpacesBefore && SpacesAfter)
1283             ++CountBoundToType;
1284         }
1285 
1286         if (Tok->WhitespaceRange.getBegin() == Tok->WhitespaceRange.getEnd()) {
1287           if (Tok->is(tok::coloncolon) && Tok->Previous->is(TT_TemplateOpener))
1288             HasCpp03IncompatibleFormat = true;
1289           if (Tok->is(TT_TemplateCloser) &&
1290               Tok->Previous->is(TT_TemplateCloser))
1291             HasCpp03IncompatibleFormat = true;
1292         }
1293 
1294         if (Tok->PackingKind == PPK_BinPacked)
1295           HasBinPackedFunction = true;
1296         if (Tok->PackingKind == PPK_OnePerLine)
1297           HasOnePerLineFunction = true;
1298 
1299         Tok = Tok->Next;
1300       }
1301     }
1302     if (Style.DerivePointerAlignment) {
1303       if (CountBoundToType > CountBoundToVariable)
1304         Style.PointerAlignment = FormatStyle::PAS_Left;
1305       else if (CountBoundToType < CountBoundToVariable)
1306         Style.PointerAlignment = FormatStyle::PAS_Right;
1307     }
1308     if (Style.Standard == FormatStyle::LS_Auto) {
1309       Style.Standard = HasCpp03IncompatibleFormat ? FormatStyle::LS_Cpp11
1310                                                   : FormatStyle::LS_Cpp03;
1311     }
1312     BinPackInconclusiveFunctions =
1313         HasBinPackedFunction || !HasOnePerLineFunction;
1314   }
1315 
1316   void consumeUnwrappedLine(const UnwrappedLine &TheLine) override {
1317     assert(!UnwrappedLines.empty());
1318     UnwrappedLines.back().push_back(TheLine);
1319   }
1320 
1321   void finishRun() override {
1322     UnwrappedLines.push_back(SmallVector<UnwrappedLine, 16>());
1323   }
1324 
1325   FormatStyle Style;
1326   FileID ID;
1327   SourceManager &SourceMgr;
1328   WhitespaceManager Whitespaces;
1329   SmallVector<CharSourceRange, 8> Ranges;
1330   SmallVector<SmallVector<UnwrappedLine, 16>, 2> UnwrappedLines;
1331 
1332   encoding::Encoding Encoding;
1333   bool BinPackInconclusiveFunctions;
1334 };
1335 
1336 } // end anonymous namespace
1337 
1338 tooling::Replacements reformat(const FormatStyle &Style, Lexer &Lex,
1339                                SourceManager &SourceMgr,
1340                                ArrayRef<CharSourceRange> Ranges) {
1341   if (Style.DisableFormat)
1342     return tooling::Replacements();
1343   return reformat(Style, SourceMgr,
1344                   SourceMgr.getFileID(Lex.getSourceLocation()), Ranges);
1345 }
1346 
1347 tooling::Replacements reformat(const FormatStyle &Style,
1348                                SourceManager &SourceMgr, FileID ID,
1349                                ArrayRef<CharSourceRange> Ranges) {
1350   if (Style.DisableFormat)
1351     return tooling::Replacements();
1352   Formatter formatter(Style, SourceMgr, ID, Ranges);
1353   return formatter.format();
1354 }
1355 
1356 tooling::Replacements reformat(const FormatStyle &Style, StringRef Code,
1357                                ArrayRef<tooling::Range> Ranges,
1358                                StringRef FileName) {
1359   if (Style.DisableFormat)
1360     return tooling::Replacements();
1361 
1362   FileManager Files((FileSystemOptions()));
1363   DiagnosticsEngine Diagnostics(
1364       IntrusiveRefCntPtr<DiagnosticIDs>(new DiagnosticIDs),
1365       new DiagnosticOptions);
1366   SourceManager SourceMgr(Diagnostics, Files);
1367   std::unique_ptr<llvm::MemoryBuffer> Buf =
1368       llvm::MemoryBuffer::getMemBuffer(Code, FileName);
1369   const clang::FileEntry *Entry =
1370       Files.getVirtualFile(FileName, Buf->getBufferSize(), 0);
1371   SourceMgr.overrideFileContents(Entry, std::move(Buf));
1372   FileID ID =
1373       SourceMgr.createFileID(Entry, SourceLocation(), clang::SrcMgr::C_User);
1374   SourceLocation StartOfFile = SourceMgr.getLocForStartOfFile(ID);
1375   std::vector<CharSourceRange> CharRanges;
1376   for (const tooling::Range &Range : Ranges) {
1377     SourceLocation Start = StartOfFile.getLocWithOffset(Range.getOffset());
1378     SourceLocation End = Start.getLocWithOffset(Range.getLength());
1379     CharRanges.push_back(CharSourceRange::getCharRange(Start, End));
1380   }
1381   return reformat(Style, SourceMgr, ID, CharRanges);
1382 }
1383 
1384 LangOptions getFormattingLangOpts(const FormatStyle &Style) {
1385   LangOptions LangOpts;
1386   LangOpts.CPlusPlus = 1;
1387   LangOpts.CPlusPlus11 = Style.Standard == FormatStyle::LS_Cpp03 ? 0 : 1;
1388   LangOpts.CPlusPlus14 = Style.Standard == FormatStyle::LS_Cpp03 ? 0 : 1;
1389   LangOpts.LineComment = 1;
1390   bool AlternativeOperators = Style.Language != FormatStyle::LK_JavaScript &&
1391                               Style.Language != FormatStyle::LK_Java;
1392   LangOpts.CXXOperatorNames = AlternativeOperators ? 1 : 0;
1393   LangOpts.Bool = 1;
1394   LangOpts.ObjC1 = 1;
1395   LangOpts.ObjC2 = 1;
1396   return LangOpts;
1397 }
1398 
1399 const char *StyleOptionHelpDescription =
1400     "Coding style, currently supports:\n"
1401     "  LLVM, Google, Chromium, Mozilla, WebKit.\n"
1402     "Use -style=file to load style configuration from\n"
1403     ".clang-format file located in one of the parent\n"
1404     "directories of the source file (or current\n"
1405     "directory for stdin).\n"
1406     "Use -style=\"{key: value, ...}\" to set specific\n"
1407     "parameters, e.g.:\n"
1408     "  -style=\"{BasedOnStyle: llvm, IndentWidth: 8}\"";
1409 
1410 static FormatStyle::LanguageKind getLanguageByFileName(StringRef FileName) {
1411   if (FileName.endswith(".java")) {
1412     return FormatStyle::LK_Java;
1413   } else if (FileName.endswith_lower(".js")) {
1414     return FormatStyle::LK_JavaScript;
1415   } else if (FileName.endswith_lower(".proto") ||
1416              FileName.endswith_lower(".protodevel")) {
1417     return FormatStyle::LK_Proto;
1418   }
1419   return FormatStyle::LK_Cpp;
1420 }
1421 
1422 FormatStyle getStyle(StringRef StyleName, StringRef FileName,
1423                      StringRef FallbackStyle) {
1424   FormatStyle Style = getLLVMStyle();
1425   Style.Language = getLanguageByFileName(FileName);
1426   if (!getPredefinedStyle(FallbackStyle, Style.Language, &Style)) {
1427     llvm::errs() << "Invalid fallback style \"" << FallbackStyle
1428                  << "\" using LLVM style\n";
1429     return Style;
1430   }
1431 
1432   if (StyleName.startswith("{")) {
1433     // Parse YAML/JSON style from the command line.
1434     if (std::error_code ec = parseConfiguration(StyleName, &Style)) {
1435       llvm::errs() << "Error parsing -style: " << ec.message() << ", using "
1436                    << FallbackStyle << " style\n";
1437     }
1438     return Style;
1439   }
1440 
1441   if (!StyleName.equals_lower("file")) {
1442     if (!getPredefinedStyle(StyleName, Style.Language, &Style))
1443       llvm::errs() << "Invalid value for -style, using " << FallbackStyle
1444                    << " style\n";
1445     return Style;
1446   }
1447 
1448   // Look for .clang-format/_clang-format file in the file's parent directories.
1449   SmallString<128> UnsuitableConfigFiles;
1450   SmallString<128> Path(FileName);
1451   llvm::sys::fs::make_absolute(Path);
1452   for (StringRef Directory = Path; !Directory.empty();
1453        Directory = llvm::sys::path::parent_path(Directory)) {
1454     if (!llvm::sys::fs::is_directory(Directory))
1455       continue;
1456     SmallString<128> ConfigFile(Directory);
1457 
1458     llvm::sys::path::append(ConfigFile, ".clang-format");
1459     DEBUG(llvm::dbgs() << "Trying " << ConfigFile << "...\n");
1460     bool IsFile = false;
1461     // Ignore errors from is_regular_file: we only need to know if we can read
1462     // the file or not.
1463     llvm::sys::fs::is_regular_file(Twine(ConfigFile), IsFile);
1464 
1465     if (!IsFile) {
1466       // Try _clang-format too, since dotfiles are not commonly used on Windows.
1467       ConfigFile = Directory;
1468       llvm::sys::path::append(ConfigFile, "_clang-format");
1469       DEBUG(llvm::dbgs() << "Trying " << ConfigFile << "...\n");
1470       llvm::sys::fs::is_regular_file(Twine(ConfigFile), IsFile);
1471     }
1472 
1473     if (IsFile) {
1474       llvm::ErrorOr<std::unique_ptr<llvm::MemoryBuffer>> Text =
1475           llvm::MemoryBuffer::getFile(ConfigFile.c_str());
1476       if (std::error_code EC = Text.getError()) {
1477         llvm::errs() << EC.message() << "\n";
1478         break;
1479       }
1480       if (std::error_code ec =
1481               parseConfiguration(Text.get()->getBuffer(), &Style)) {
1482         if (ec == ParseError::Unsuitable) {
1483           if (!UnsuitableConfigFiles.empty())
1484             UnsuitableConfigFiles.append(", ");
1485           UnsuitableConfigFiles.append(ConfigFile);
1486           continue;
1487         }
1488         llvm::errs() << "Error reading " << ConfigFile << ": " << ec.message()
1489                      << "\n";
1490         break;
1491       }
1492       DEBUG(llvm::dbgs() << "Using configuration file " << ConfigFile << "\n");
1493       return Style;
1494     }
1495   }
1496   llvm::errs() << "Can't find usable .clang-format, using " << FallbackStyle
1497                << " style\n";
1498   if (!UnsuitableConfigFiles.empty()) {
1499     llvm::errs() << "Configuration file(s) do(es) not support "
1500                  << getLanguageName(Style.Language) << ": "
1501                  << UnsuitableConfigFiles << "\n";
1502   }
1503   return Style;
1504 }
1505 
1506 } // namespace format
1507 } // namespace clang
1508