mirror of
https://github.com/carbon-language/carbon-lang.git
synced 2026-10-05 22:02:55 +01:00
Makes following changes to Carbon::Format() - TokenKind::Period (i.e. `.`) should never have a space before or after it. - TokenKind::CloseSquareParen (i.e. `]`) should be treated as packed content (no space preceeding it) - Only exception I can think of is `impl forall [...]` - Remove preceeding space from `[` and `(` if previous token was an identifier (or identifier-ish token) - Remove seperator following `++` / `--` unary operators. - Explicit gaps in source code should be retained, up to 2 new lines. Multiple test files were added to test formatting. I imagine eventually this will need to be updated to read parse tree to gather more context but this atleast lets us get a decent-ish format for many of our current sample files (e.g. sieve.carbon) Assisted-With: Gemini / Antigravity --------- Co-authored-by: David Blaikie <dblaikie@gmail.com>
115 lines
4.2 KiB
C++
115 lines
4.2 KiB
C++
// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
|
|
// Exceptions. See /LICENSE for license information.
|
|
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
|
|
|
#ifndef CARBON_TOOLCHAIN_FORMAT_FORMATTER_H_
|
|
#define CARBON_TOOLCHAIN_FORMAT_FORMATTER_H_
|
|
|
|
#include <cstdint>
|
|
|
|
#include "common/ostream.h"
|
|
#include "toolchain/lex/tokenized_buffer.h"
|
|
|
|
namespace Carbon::Format {
|
|
|
|
// Implements Format(); see format.h. It's intended to be constructed and
|
|
// `Run()` once, then destructed.
|
|
//
|
|
// TODO: This will probably need to work less linearly in the future, for
|
|
// example to handle smart wrapping of arguments. This is a simple
|
|
// implementation that only handles simple code. Before adding too much more
|
|
// complexity, it should be rewritten.
|
|
//
|
|
// TODO: Add retention of blank lines between original code.
|
|
//
|
|
// TODO: Add support for formatting line ranges (will need flags too).
|
|
class Formatter {
|
|
public:
|
|
explicit Formatter(const Lex::TokenizedBuffer* tokens, llvm::raw_ostream* out)
|
|
: tokens_(tokens),
|
|
out_(out),
|
|
next_comment_(tokens->comments().begin()),
|
|
comments_end_(tokens->comments().end()) {}
|
|
|
|
// See class comments.
|
|
auto Run() -> bool;
|
|
|
|
private:
|
|
// Tracks the status of the current line of output.
|
|
enum class LineState : uint8_t {
|
|
// There is no output for the current line.
|
|
Empty,
|
|
// The current line has content (possibly just an indent), and does not need
|
|
// a separator added.
|
|
HasSeparator,
|
|
// The current line has content, and will need a separator, typically a
|
|
// single space or newline.
|
|
NeedsSeparator,
|
|
// The current line has content and is complete; a newline is pending but
|
|
// has not yet been emitted. We defer the newline so that a trailing comment
|
|
// can still be attached to this line before it is broken. The newline is
|
|
// materialized by the next content emitted (see `PrepareForPackedContent`)
|
|
// or when the file ends.
|
|
EndOfLine,
|
|
};
|
|
|
|
// Marks the current line as complete, so the next content starts a new line.
|
|
// The newline is deferred rather than emitted immediately, allowing a
|
|
// trailing comment to be attached first. Does not indent, allowing blank
|
|
// lines.
|
|
auto RequireEmptyLine() -> void;
|
|
|
|
// Emits the comment at `next_comment_` and advances past it. A trailing
|
|
// comment is kept on the current line (separated by a space) when there is
|
|
// still content to attach it to; otherwise the comment is emitted on its own
|
|
// line.
|
|
auto EmitComment() -> void;
|
|
|
|
// Emits a new line before next token. If the source code contained multiple
|
|
// newlines, will emit up to 2 new lines.
|
|
auto EmitNewLine(int start_line) -> void;
|
|
|
|
// Ensures there is a separator before adding new content. May do
|
|
// `PrepareForPackedContent` or output a separator space, dependent on line
|
|
// state. Always results in line_state_ being HasSeparator; the caller is
|
|
// responsible for adjusting state if needed.
|
|
auto PrepareForSpacedContent(int start_line = 0) -> void;
|
|
|
|
// Requires that the current line is indented, but not necessarily a separator
|
|
// space. May output spaces for `indent_`, dependent on line state. Only
|
|
// guarantees the line_state_ is not Empty; the caller is responsible for
|
|
// adjusting state if needed.
|
|
auto PrepareForPackedContent(int start_line = 0) -> void;
|
|
|
|
// Returns the next token index.
|
|
static auto NextToken(Lex::TokenIndex token) -> Lex::TokenIndex {
|
|
return *(Lex::TokenIterator(token) + 1);
|
|
}
|
|
|
|
// The tokens being formatted.
|
|
const Lex::TokenizedBuffer* tokens_;
|
|
|
|
// The output stream for formatted content.
|
|
llvm::raw_ostream* out_;
|
|
|
|
// The next comment to emit, and one past the last comment.
|
|
Lex::CommentIterator next_comment_;
|
|
Lex::CommentIterator comments_end_;
|
|
|
|
// The state of the line currently written to output.
|
|
LineState line_state_ = LineState::Empty;
|
|
|
|
// The current code indent level, to be added to new lines.
|
|
int indent_ = 0;
|
|
|
|
// The 0-based end line in original source of the previous token or comment.
|
|
int prev_end_line_ = 0;
|
|
|
|
// Kind of the last token before current one.
|
|
Lex::TokenKind prev_token_kind_ = Lex::TokenKind::FileStart;
|
|
};
|
|
|
|
} // namespace Carbon::Format
|
|
|
|
#endif // CARBON_TOOLCHAIN_FORMAT_FORMATTER_H_
|