llvm-project/clang/lib/AST/CommentBriefParser.cpp

//===--- CommentBriefParser.cpp - Dumb comment parser ---------------------===//
//
//                     The LLVM Compiler Infrastructure
//
// This file is distributed under the University of Illinois Open Source
// License. See LICENSE.TXT for details.
//
//===----------------------------------------------------------------------===//

#include "clang/AST/CommentBriefParser.h"
#include "clang/AST/CommentCommandTraits.h"
#include "llvm/ADT/StringSwitch.h"

namespace clang {
namespace comments {

namespace {
inline bool isWhitespace(char C) {
  return C == ' ' || C == '\n' || C == '\r' ||
         C == '\t' || C == '\f' || C == '\v';
}

/// Convert all whitespace into spaces, remove leading and trailing spaces,
/// compress multiple spaces into one.
void cleanupBrief(std::string &S) {
  bool PrevWasSpace = true;
  std::string::iterator O = S.begin();
  for (std::string::iterator I = S.begin(), E = S.end();
       I != E; ++I) {
    const char C = *I;
    if (isWhitespace(C)) {
      if (!PrevWasSpace) {
        *O++ = ' ';
        PrevWasSpace = true;
      }
      continue;
    } else {
      *O++ = C;
      PrevWasSpace = false;
    }
  }
  if (O != S.begin() && *(O - 1) == ' ')
    --O;

  S.resize(O - S.begin());
}

bool isWhitespace(StringRef Text) {
  for (StringRef::const_iterator I = Text.begin(), E = Text.end();
       I != E; ++I) {
    if (!isWhitespace(*I))
      return false;
  }
  return true;
}
} // unnamed namespace

BriefParser::BriefParser(Lexer &L, const CommandTraits &Traits) :
    L(L), Traits(Traits) {
  // Get lookahead token.
  ConsumeToken();
}

std::string BriefParser::Parse() {
  std::string FirstParagraphOrBrief;
  std::string ReturnsParagraph;
  bool InFirstParagraph = true;
  bool InBrief = false;
  bool InReturns = false;

  while (Tok.isNot(tok::eof)) {
    if (Tok.is(tok::text)) {
      if (InFirstParagraph || InBrief)
        FirstParagraphOrBrief += Tok.getText();
      else if (InReturns)
        ReturnsParagraph += Tok.getText();
      ConsumeToken();
      continue;
    }

    if (Tok.is(tok::backslash_command) || Tok.is(tok::at_command)) {
      const CommandInfo *Info = Traits.getCommandInfo(Tok.getCommandID());
      if (Info->IsBriefCommand) {
        FirstParagraphOrBrief.clear();
        InBrief = true;
        ConsumeToken();
        continue;
      }
      if (Info->IsReturnsCommand) {
        InReturns = true;
        InBrief = false;
        InFirstParagraph = false;
        ReturnsParagraph += "Returns ";
        ConsumeToken();
        continue;
      }
      // Block commands implicitly start a new paragraph.
      if (Info->IsBlockCommand) {
        // We found an implicit paragraph end.
        InFirstParagraph = false;
        if (InBrief)
          break;
      }
    }

    if (Tok.is(tok::newline)) {
      if (InFirstParagraph || InBrief)
        FirstParagraphOrBrief += ' ';
      else if (InReturns)
        ReturnsParagraph += ' ';
      ConsumeToken();

      // If the next token is a whitespace only text, ignore it.  Thus we allow
      // two paragraphs to be separated by line that has only whitespace in it.
      //
      // We don't need to add a space to the parsed text because we just added
      // a space for the newline.
      if (Tok.is(tok::text)) {
        if (isWhitespace(Tok.getText()))
          ConsumeToken();
      }

      if (Tok.is(tok::newline)) {
        ConsumeToken();
        // We found a paragraph end.  This ends the brief description if
        // \\brief command or its equivalent was explicitly used.
        // Stop scanning text because an explicit \\brief paragraph is the
        // preffered one.
        if (InBrief)
          break;
        // End first paragraph if we found some non-whitespace text.
        if (InFirstParagraph && !isWhitespace(FirstParagraphOrBrief))
          InFirstParagraph = false;
        // End the \\returns paragraph because we found the paragraph end.
        InReturns = false;
      }
      continue;
    }

    // We didn't handle this token, so just drop it.
    ConsumeToken();
  }

  cleanupBrief(FirstParagraphOrBrief);
  if (!FirstParagraphOrBrief.empty())
    return FirstParagraphOrBrief;

  cleanupBrief(ReturnsParagraph);
  return ReturnsParagraph;
}

} // end namespace comments
} // end namespace clang
Implement a lexer for structured comments. llvm-svn: 159223 2012-06-27 04:39:18 +08:00			`//===--- CommentBriefParser.cpp - Dumb comment parser ---------------------===//`
			`//`
			`// The LLVM Compiler Infrastructure`
			`//`
			`// This file is distributed under the University of Illinois Open Source`
			`// License. See LICENSE.TXT for details.`
			`//`
			`//===----------------------------------------------------------------------===//`

			`#include "clang/AST/CommentBriefParser.h"`
Comment parsing: extract TableGen'able pieces into new CommandTraits class. llvm-svn: 161548 2012-08-09 08:03:17 +08:00			`#include "clang/AST/CommentCommandTraits.h"`
Factor out a check for block commands (that implicitly start a new paragraph) into a separate function. llvm-svn: 159444 2012-06-30 02:19:20 +08:00			`#include "llvm/ADT/StringSwitch.h"`
Implement a lexer for structured comments. llvm-svn: 159223 2012-06-27 04:39:18 +08:00
			`namespace clang {`
			`namespace comments {`

Cleanup \brief comment. Since it is a single paragraph, no need to save newlines there. llvm-svn: 159325 2012-06-28 09:38:21 +08:00			`namespace {`
CommentBriefParser: allow paragraphs to be separated by line of whitespace. Skip paragraphs that contain only whitespace. llvm-svn: 162315 2012-08-22 05:15:34 +08:00			`inline bool isWhitespace(char C) {`
			`return C == ' ' \|\| C == '\n' \|\| C == '\r' \|\|`
			`C == '\t' \|\| C == '\f' \|\| C == '\v';`
			`}`

Cleanup \brief comment. Since it is a single paragraph, no need to save newlines there. llvm-svn: 159325 2012-06-28 09:38:21 +08:00			`/// Convert all whitespace into spaces, remove leading and trailing spaces,`
			`/// compress multiple spaces into one.`
			`void cleanupBrief(std::string &S) {`
			`bool PrevWasSpace = true;`
			`std::string::iterator O = S.begin();`
			`for (std::string::iterator I = S.begin(), E = S.end();`
			`I != E; ++I) {`
			`const char C = *I;`
CommentBriefParser: allow paragraphs to be separated by line of whitespace. Skip paragraphs that contain only whitespace. llvm-svn: 162315 2012-08-22 05:15:34 +08:00			`if (isWhitespace(C)) {`
Cleanup \brief comment. Since it is a single paragraph, no need to save newlines there. llvm-svn: 159325 2012-06-28 09:38:21 +08:00			`if (!PrevWasSpace) {`
			`*O++ = ' ';`
			`PrevWasSpace = true;`
			`}`
			`continue;`
			`} else {`
			`*O++ = C;`
			`PrevWasSpace = false;`
			`}`
			`}`
			`if (O != S.begin() && *(O - 1) == ' ')`
			`--O;`

			`S.resize(O - S.begin());`
			`}`
CommentBriefParser: allow paragraphs to be separated by line of whitespace. Skip paragraphs that contain only whitespace. llvm-svn: 162315 2012-08-22 05:15:34 +08:00
			`bool isWhitespace(StringRef Text) {`
			`for (StringRef::const_iterator I = Text.begin(), E = Text.end();`
			`I != E; ++I) {`
			`if (!isWhitespace(*I))`
			`return false;`
			`}`
			`return true;`
			`}`
Comment parsing: extract TableGen'able pieces into new CommandTraits class. llvm-svn: 161548 2012-08-09 08:03:17 +08:00			`} // unnamed namespace`
Factor out a check for block commands (that implicitly start a new paragraph) into a separate function. llvm-svn: 159444 2012-06-30 02:19:20 +08:00
Comment parsing: extract TableGen'able pieces into new CommandTraits class. llvm-svn: 161548 2012-08-09 08:03:17 +08:00			`BriefParser::BriefParser(Lexer &L, const CommandTraits &Traits) :`
			`L(L), Traits(Traits) {`
			`// Get lookahead token.`
			`ConsumeToken();`
Factor out a check for block commands (that implicitly start a new paragraph) into a separate function. llvm-svn: 159444 2012-06-30 02:19:20 +08:00			`}`
Cleanup \brief comment. Since it is a single paragraph, no need to save newlines there. llvm-svn: 159325 2012-06-28 09:38:21 +08:00
Implement a lexer for structured comments. llvm-svn: 159223 2012-06-27 04:39:18 +08:00			`std::string BriefParser::Parse() {`
CommentBriefParser: use \returns if we can't find the \brief or just a plain paragraph. llvm-svn: 160550 2012-07-21 01:01:34 +08:00			`std::string FirstParagraphOrBrief;`
			`std::string ReturnsParagraph;`
Implement a lexer for structured comments. llvm-svn: 159223 2012-06-27 04:39:18 +08:00			`bool InFirstParagraph = true;`
			`bool InBrief = false;`
CommentBriefParser: use \returns if we can't find the \brief or just a plain paragraph. llvm-svn: 160550 2012-07-21 01:01:34 +08:00			`bool InReturns = false;`
Implement a lexer for structured comments. llvm-svn: 159223 2012-06-27 04:39:18 +08:00
			`while (Tok.isNot(tok::eof)) {`
			`if (Tok.is(tok::text)) {`
Simplify logic in BriefParser::Parse(), per Jordan's comment. llvm-svn: 159247 2012-06-27 09:17:34 +08:00			`if (InFirstParagraph \|\| InBrief)`
CommentBriefParser: use \returns if we can't find the \brief or just a plain paragraph. llvm-svn: 160550 2012-07-21 01:01:34 +08:00			`FirstParagraphOrBrief += Tok.getText();`
			`else if (InReturns)`
			`ReturnsParagraph += Tok.getText();`
Implement a lexer for structured comments. llvm-svn: 159223 2012-06-27 04:39:18 +08:00			`ConsumeToken();`
			`continue;`
			`}`

Some refactoring in my patch on document command source fidelity. // rdar://13066276 llvm-svn: 176401 2013-03-02 10:39:57 +08:00			`if (Tok.is(tok::backslash_command) \|\| Tok.is(tok::at_command)) {`
Comment AST: TableGen'ize all command lists in CommentCommandTraits.cpp. Now we have a list of all commands. This is a good thing in itself, but it also enables us to easily implement typo correction for command names. With this change we have objects that contain information about each command, so it makes sense to resolve command name just once during lexing (currently we store command names as strings and do a linear search every time some property value is needed). Thus comment token and AST nodes were changed to contain a command ID -- index into a tables of builtin and registered commands. Unknown commands are registered during parsing and thus are also uniformly assigned an ID. Using an ID instead of a StringRef is also a nice memory optimization since ID is a small integer that fits into a common bitfield in Comment class. This change implies that to get any information about a command (even a command name) we need a CommandTraits object to resolve the command ID to CommandInfo*. Currently a fresh temporary CommandTraits object is created whenever it is needed since it does not have any state. But with this change it has state -- new commands can be registered, so a CommandTraits object was added to ASTContext. Also, in libclang CXComment has to be expanded to include a CXTranslationUnit so that all functions working on comment AST nodes can get a CommandTraits object. This breaks binary compatibility of CXComment APIs. Now clang_FullComment_getAsXML(CXTranslationUnit TU, CXComment CXC) doesn't need TU parameter anymore, so it was removed. This is a source-incompatible change for this C API. llvm-svn: 163540 2012-09-11 04:32:42 +08:00			`const CommandInfo *Info = Traits.getCommandInfo(Tok.getCommandID());`
			`if (Info->IsBriefCommand) {`
CommentBriefParser: use \returns if we can't find the \brief or just a plain paragraph. llvm-svn: 160550 2012-07-21 01:01:34 +08:00			`FirstParagraphOrBrief.clear();`
Teach \brief parser about commands that start a new paragraph implicitly llvm-svn: 159309 2012-06-28 08:01:41 +08:00			`InBrief = true;`
			`ConsumeToken();`
			`continue;`
			`}`
Comment AST: TableGen'ize all command lists in CommentCommandTraits.cpp. Now we have a list of all commands. This is a good thing in itself, but it also enables us to easily implement typo correction for command names. With this change we have objects that contain information about each command, so it makes sense to resolve command name just once during lexing (currently we store command names as strings and do a linear search every time some property value is needed). Thus comment token and AST nodes were changed to contain a command ID -- index into a tables of builtin and registered commands. Unknown commands are registered during parsing and thus are also uniformly assigned an ID. Using an ID instead of a StringRef is also a nice memory optimization since ID is a small integer that fits into a common bitfield in Comment class. This change implies that to get any information about a command (even a command name) we need a CommandTraits object to resolve the command ID to CommandInfo*. Currently a fresh temporary CommandTraits object is created whenever it is needed since it does not have any state. But with this change it has state -- new commands can be registered, so a CommandTraits object was added to ASTContext. Also, in libclang CXComment has to be expanded to include a CXTranslationUnit so that all functions working on comment AST nodes can get a CommandTraits object. This breaks binary compatibility of CXComment APIs. Now clang_FullComment_getAsXML(CXTranslationUnit TU, CXComment CXC) doesn't need TU parameter anymore, so it was removed. This is a source-incompatible change for this C API. llvm-svn: 163540 2012-09-11 04:32:42 +08:00			`if (Info->IsReturnsCommand) {`
CommentBriefParser: use \returns if we can't find the \brief or just a plain paragraph. llvm-svn: 160550 2012-07-21 01:01:34 +08:00			`InReturns = true;`
CommentBriefParser: allow paragraphs to be separated by line of whitespace. Skip paragraphs that contain only whitespace. llvm-svn: 162315 2012-08-22 05:15:34 +08:00			`InBrief = false;`
			`InFirstParagraph = false;`
CommentBriefParser: use \returns if we can't find the \brief or just a plain paragraph. llvm-svn: 160550 2012-07-21 01:01:34 +08:00			`ReturnsParagraph += "Returns ";`
CommentBriefParser: allow paragraphs to be separated by line of whitespace. Skip paragraphs that contain only whitespace. llvm-svn: 162315 2012-08-22 05:15:34 +08:00			`ConsumeToken();`
			`continue;`
CommentBriefParser: use \returns if we can't find the \brief or just a plain paragraph. llvm-svn: 160550 2012-07-21 01:01:34 +08:00			`}`
Factor out a check for block commands (that implicitly start a new paragraph) into a separate function. llvm-svn: 159444 2012-06-30 02:19:20 +08:00			`// Block commands implicitly start a new paragraph.`
Comment AST: TableGen'ize all command lists in CommentCommandTraits.cpp. Now we have a list of all commands. This is a good thing in itself, but it also enables us to easily implement typo correction for command names. With this change we have objects that contain information about each command, so it makes sense to resolve command name just once during lexing (currently we store command names as strings and do a linear search every time some property value is needed). Thus comment token and AST nodes were changed to contain a command ID -- index into a tables of builtin and registered commands. Unknown commands are registered during parsing and thus are also uniformly assigned an ID. Using an ID instead of a StringRef is also a nice memory optimization since ID is a small integer that fits into a common bitfield in Comment class. This change implies that to get any information about a command (even a command name) we need a CommandTraits object to resolve the command ID to CommandInfo*. Currently a fresh temporary CommandTraits object is created whenever it is needed since it does not have any state. But with this change it has state -- new commands can be registered, so a CommandTraits object was added to ASTContext. Also, in libclang CXComment has to be expanded to include a CXTranslationUnit so that all functions working on comment AST nodes can get a CommandTraits object. This breaks binary compatibility of CXComment APIs. Now clang_FullComment_getAsXML(CXTranslationUnit TU, CXComment CXC) doesn't need TU parameter anymore, so it was removed. This is a source-incompatible change for this C API. llvm-svn: 163540 2012-09-11 04:32:42 +08:00			`if (Info->IsBlockCommand) {`
Teach \brief parser about commands that start a new paragraph implicitly llvm-svn: 159309 2012-06-28 08:01:41 +08:00			`// We found an implicit paragraph end.`
			`InFirstParagraph = false;`
CommentBriefParser: remove dead store. Found by Clang Analyzer. llvm-svn: 159673 2012-07-04 02:10:20 +08:00			`if (InBrief)`
Teach \brief parser about commands that start a new paragraph implicitly llvm-svn: 159309 2012-06-28 08:01:41 +08:00			`break;`
			`}`
Implement a lexer for structured comments. llvm-svn: 159223 2012-06-27 04:39:18 +08:00			`}`

			`if (Tok.is(tok::newline)) {`
Simplify logic in BriefParser::Parse(), per Jordan's comment. llvm-svn: 159247 2012-06-27 09:17:34 +08:00			`if (InFirstParagraph \|\| InBrief)`
CommentBriefParser: use \returns if we can't find the \brief or just a plain paragraph. llvm-svn: 160550 2012-07-21 01:01:34 +08:00			`FirstParagraphOrBrief += ' ';`
			`else if (InReturns)`
			`ReturnsParagraph += ' ';`
Implement a lexer for structured comments. llvm-svn: 159223 2012-06-27 04:39:18 +08:00			`ConsumeToken();`

CommentBriefParser: allow paragraphs to be separated by line of whitespace. Skip paragraphs that contain only whitespace. llvm-svn: 162315 2012-08-22 05:15:34 +08:00			`// If the next token is a whitespace only text, ignore it. Thus we allow`
			`// two paragraphs to be separated by line that has only whitespace in it.`
			`//`
			`// We don't need to add a space to the parsed text because we just added`
			`// a space for the newline.`
			`if (Tok.is(tok::text)) {`
			`if (isWhitespace(Tok.getText()))`
			`ConsumeToken();`
			`}`

Implement a lexer for structured comments. llvm-svn: 159223 2012-06-27 04:39:18 +08:00			`if (Tok.is(tok::newline)) {`
			`ConsumeToken();`
CommentBriefParser: allow paragraphs to be separated by line of whitespace. Skip paragraphs that contain only whitespace. llvm-svn: 162315 2012-08-22 05:15:34 +08:00			`// We found a paragraph end. This ends the brief description if`
			`// \\brief command or its equivalent was explicitly used.`
			`// Stop scanning text because an explicit \\brief paragraph is the`
			`// preffered one.`
CommentBriefParser: remove dead store. Found by Clang Analyzer. llvm-svn: 159673 2012-07-04 02:10:20 +08:00			`if (InBrief)`
Teach \brief parser about commands that start a new paragraph implicitly llvm-svn: 159309 2012-06-28 08:01:41 +08:00			`break;`
CommentBriefParser: allow paragraphs to be separated by line of whitespace. Skip paragraphs that contain only whitespace. llvm-svn: 162315 2012-08-22 05:15:34 +08:00			`// End first paragraph if we found some non-whitespace text.`
			`if (InFirstParagraph && !isWhitespace(FirstParagraphOrBrief))`
			`InFirstParagraph = false;`
			`// End the \\returns paragraph because we found the paragraph end.`
			`InReturns = false;`
Implement a lexer for structured comments. llvm-svn: 159223 2012-06-27 04:39:18 +08:00			`}`
			`continue;`
			`}`

			`// We didn't handle this token, so just drop it.`
			`ConsumeToken();`
			`}`

CommentBriefParser: use \returns if we can't find the \brief or just a plain paragraph. llvm-svn: 160550 2012-07-21 01:01:34 +08:00			`cleanupBrief(FirstParagraphOrBrief);`
			`if (!FirstParagraphOrBrief.empty())`
			`return FirstParagraphOrBrief;`

			`cleanupBrief(ReturnsParagraph);`
			`return ReturnsParagraph;`
Implement a lexer for structured comments. llvm-svn: 159223 2012-06-27 04:39:18 +08:00			`}`

			`} // end namespace comments`
			`} // end namespace clang`