Factor out FormatterChunks logic (#6779)

I'm looking at making `constants { ... }` etc omitted when empty,
because in turn I'm looking at adding a third section, and seeing more
boilerplate empty sections just seems awkward to me. This PR starts down
the path by factoring out the chunk logic, which I may want to refactor
further.

This changes the `size_t` chunk id into a wrapped type for type safety.

This PR is just a refactoring, and doesn't make any behavior changes.

Assisted-by: Google Antigravity with Gemini 3 Flash
This commit is contained in:
Jon Ross-Perkins
2026-02-24 19:07:31 +00:00
committed by GitHub
parent 8edd5eb9a1
commit 1a3f762dba
5 changed files with 425 additions and 342 deletions
+11
View File
@@ -210,6 +210,16 @@ cc_library(
],
)
cc_library(
name = "formatter_chunks",
srcs = ["formatter_chunks.cpp"],
hdrs = ["formatter_chunks.h"],
deps = [
"//common:check",
"@llvm-project//llvm:Support",
],
)
cc_library(
name = "formatter",
srcs = ["formatter.cpp"],
@@ -217,6 +227,7 @@ cc_library(
deps = [
":expr_info",
":file",
":formatter_chunks",
":inst_namer",
":typed_insts",
"//common:concepts",
File diff suppressed because it is too large Load Diff
+29 -83
View File
@@ -12,6 +12,7 @@
#include "toolchain/base/fixed_size_value_store.h"
#include "toolchain/parse/tree_and_subtrees.h"
#include "toolchain/sem_ir/file.h"
#include "toolchain/sem_ir/formatter_chunks.h"
#include "toolchain/sem_ir/inst_namer.h"
namespace Carbon::SemIR {
@@ -37,15 +38,15 @@ class Formatter {
// - Entities are difficult to order (forward declarations may lead to
// circular references), and so are simply grouped by type.
//
// When formatting constants and imports, we use `OutputChunks` to only print
// entities which are referenced. For example, imports speculatively create
// constants which may never be referenced, or for which the referencing
// instruction may be hidden and we normally hide those. See `OutputChunk` for
// additional information.
// When formatting constants and imports, we use `FormatterChunks` to only
// print entities which are referenced. For example, imports speculatively
// create constants which may never be referenced, or for which the
// referencing instruction may be hidden and we normally hide those. See
// `FormatterChunks` for additional information.
//
// Beyond `OutputChunk`, `ShouldFormatEntity` and `ShouldFormatInst` can also
// hide instructions. These interact because an hidden instruction means its
// references are unused for `OutputChunk` visibility.
// Beyond `FormatterChunks`, `ShouldFormatEntity` and `ShouldFormatInst` can
// also hide instructions. These interact because an hidden instruction means
// its references are unused for `FormatterChunks` visibility.
auto Format() -> void;
// Write buffered output to the given stream. `Format` must be called first.
@@ -54,63 +55,10 @@ class Formatter {
private:
enum class AddSpace : bool { Before, After };
// A chunk of the buffered output. Constants and imports are buffered as
// `OutputChunk`s until we reach the end of formatting so that we can decide
// whether to include them based on whether they are referenced.
//
// When `FormatName` is called for an instruction, it's considered referenced;
// if that instruction is in an `OutputChunk`, it and all of its dependencies
// will be marked for printing by `Write`. If that doesn't occur by the end,
// it will be omitted.
struct OutputChunk {
// Whether this chunk is known to be included in the output.
bool include_in_output;
// The textual contents of this chunk.
std::string chunk = std::string();
// Indices in `ouput_chunks_` that should be included in the output if this
// one is.
llvm::SmallVector<size_t> dependencies = {};
};
// All formatted output within the scope of this object is redirected to a
// new tentative `OutputChunk`. The new chunk will depend on
// `parent_chunk_index`.
struct TentativeOutputScope {
explicit TentativeOutputScope(Formatter& f, size_t parent_chunk_index)
: formatter(f) {
// If our parent is not known to be included, create a new chunk and
// include it only if the parent is later found to be used.
if (!f.output_chunks_[parent_chunk_index].include_in_output) {
index = formatter.AddChunk(false);
f.output_chunks_[parent_chunk_index].dependencies.push_back(index);
}
}
~TentativeOutputScope() {
auto next_index = formatter.AddChunk(true);
CARBON_CHECK(next_index == index + 1, "Nested TentativeOutputScope");
}
Formatter& formatter;
size_t index;
};
// Fills `node_parents_` with parent information. Called at most once during
// construction.
auto ComputeNodeParents() -> void;
// Flushes the buffered output to the current chunk.
auto FlushChunk() -> void;
// Adds a new chunk to the output. Does not flush existing output, so should
// only be called if there is no buffered output.
auto AddChunkNoFlush(bool include_in_output) -> size_t;
// Flushes the current chunk and add a new chunk to the output.
auto AddChunk(bool include_in_output) -> size_t;
// Marks the given chunk as being included in the output if the current chunk
// is.
auto IncludeChunkInOutput(size_t chunk) -> void;
// Returns true if the instruction should be included according to its
// originating IR. Typically `ShouldFormatEntity` should be used instead.
auto ShouldIncludeInstByIR(InstId inst_id) -> bool;
@@ -286,9 +234,9 @@ class Formatter {
template <typename... Args>
auto FormatArgs(Args... args) -> void {
out_ << ' ';
out() << ' ';
llvm::ListSeparator sep;
((out_ << sep, FormatArg(args)), ...);
((out() << sep, FormatArg(args)), ...);
}
// FormatArg variants handling printing instruction arguments. Several things
@@ -303,16 +251,16 @@ class Formatter {
FormatName(id);
}
auto FormatArg(BoolValue v) -> void { out_ << v; }
auto FormatArg(CharId c) -> void { out_ << c; }
auto FormatArg(BoolValue v) -> void { out() << v; }
auto FormatArg(CharId c) -> void { out() << c; }
auto FormatArg(EntityNameId id) -> void;
auto FormatArg(FacetTypeId id) -> void;
auto FormatArg(IntKind k) -> void { k.Print(out_); }
auto FormatArg(FloatKind k) -> void { k.Print(out_); }
auto FormatArg(IntKind k) -> void { k.Print(out()); }
auto FormatArg(FloatKind k) -> void { k.Print(out()); }
auto FormatArg(ImportIRId id) -> void;
auto FormatArg(IntId id) -> void;
auto FormatArg(ElementIndex index) -> void { out_ << index; }
auto FormatArg(CallParamIndex index) -> void { out_ << index; }
auto FormatArg(ElementIndex index) -> void { out() << index; }
auto FormatArg(CallParamIndex index) -> void { out() << index; }
auto FormatArg(NameScopeId id) -> void;
auto FormatArg(InstBlockId id) -> void;
auto FormatArg(AbsoluteInstBlockId id) -> void;
@@ -338,7 +286,7 @@ class Formatter {
requires(InstNamer::ScopeIdTypeEnum::Contains<IdT> ||
std::same_as<IdT, GenericId>)
auto FormatName(IdT id) -> void {
out_ << inst_namer_.GetNameFor(id);
out() << inst_namer_.GetNameFor(id);
}
auto FormatName(NameId id) -> void;
@@ -357,6 +305,12 @@ class Formatter {
// Returns the label for the indicated IR.
auto GetImportIRLabel(ImportIRId id) -> std::string;
// The output stream.
auto out() -> llvm::raw_ostream& { return chunks_.out(); }
// The output chunks.
FormatterChunks chunks_;
const File* sem_ir_;
InstNamer inst_namer_;
Parse::GetTreeAndSubtreesFn get_tree_and_subtrees_;
@@ -367,15 +321,6 @@ class Formatter {
// Whether to use ranges when dumping, or to dump the full SemIR.
bool use_dump_sem_ir_ranges_;
// The output stream buffer.
std::string buffer_;
// The output stream.
llvm::raw_string_ostream out_ = llvm::raw_string_ostream(buffer_);
// Chunks of output text that we have created so far.
llvm::SmallVector<OutputChunk> output_chunks_;
// The current scope that we are formatting within. References to names in
// this scope will not have a `@scope.` prefix added.
InstNamer::ScopeId scope_ = InstNamer::ScopeId::None;
@@ -409,7 +354,8 @@ class Formatter {
// Indexes of chunks of output that should be included when an instruction is
// referenced, indexed by the instruction's index.
FixedSizeValueStore<InstId, size_t, Tag<CheckIRId>> tentative_inst_chunks_;
FixedSizeValueStore<InstId, FormatterChunks::ChunkId, Tag<CheckIRId>>
tentative_inst_chunks_;
// Maps nodes to their parents. Only set when dump ranges are in use, because
// the parents aren't used otherwise.
@@ -424,15 +370,15 @@ auto Formatter::FormatEntityStart(llvm::StringRef entity_kind,
FormatGenericStart(entity_kind, generic_id);
}
out_ << "\n";
out() << "\n";
after_open_brace_ = false;
Indent();
out_ << entity_kind;
out() << entity_kind;
// If there's a generic, it will have attached the name. Otherwise, add the
// name here.
if (!generic_id.has_value()) {
out_ << " ";
out() << " ";
FormatName(entity_id);
}
}
+76
View File
@@ -0,0 +1,76 @@
// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
// Exceptions. See /LICENSE for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#include "toolchain/sem_ir/formatter_chunks.h"
#include "common/check.h"
namespace Carbon::SemIR {
FormatterChunks::TentativeScope::TentativeScope(FormatterChunks* chunks,
ChunkId parent_chunk)
: chunks(chunks) {
// If our parent is not known to be included, create a new chunk and
// include it only if the parent is later found to be used.
if (!chunks->output_chunks_[parent_chunk.index].include_in_output) {
chunk = chunks->AddChunk(false);
chunks->output_chunks_[parent_chunk.index].dependencies.push_back(chunk);
}
}
FormatterChunks::TentativeScope::~TentativeScope() {
auto next_chunk = chunks->AddChunk(true);
CARBON_CHECK(next_chunk.index == chunk.index + 1,
"Nested FormatterChunks::TentativeScope");
}
auto FormatterChunks::FlushChunk() -> void {
CARBON_CHECK(output_chunks_.back().chunk.empty());
output_chunks_.back().chunk = std::move(buffer_);
buffer_.clear();
}
auto FormatterChunks::AddChunkNoFlush(bool include_in_output) -> ChunkId {
CARBON_CHECK(buffer_.empty());
output_chunks_.push_back({.include_in_output = include_in_output});
return ChunkId{.index = output_chunks_.size() - 1};
}
auto FormatterChunks::AddChunk(bool include_in_output) -> ChunkId {
FlushChunk();
return AddChunkNoFlush(include_in_output);
}
auto FormatterChunks::IncludeChunkInOutput(ChunkId chunk) -> void {
CARBON_CHECK(chunk.index != output_chunks_.size() - 1,
"Should only be called on earlier chunks");
if (auto& current_chunk = output_chunks_.back();
!current_chunk.include_in_output) {
current_chunk.dependencies.push_back(chunk);
return;
}
llvm::SmallVector<ChunkId> to_add = {chunk};
while (!to_add.empty()) {
auto& chunk_ref = output_chunks_[to_add.pop_back_val().index];
if (chunk_ref.include_in_output) {
continue;
}
chunk_ref.include_in_output = true;
to_add.append(chunk_ref.dependencies);
chunk_ref.dependencies.clear();
}
}
auto FormatterChunks::Write(llvm::raw_ostream& stream) -> void {
FlushChunk();
for (const auto& chunk : output_chunks_) {
if (chunk.include_in_output) {
stream << chunk.chunk;
}
}
}
} // namespace Carbon::SemIR
+87
View File
@@ -0,0 +1,87 @@
// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
// Exceptions. See /LICENSE for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#ifndef CARBON_TOOLCHAIN_SEM_IR_FORMATTER_CHUNKS_H_
#define CARBON_TOOLCHAIN_SEM_IR_FORMATTER_CHUNKS_H_
#include <string>
#include "llvm/ADT/SmallVector.h"
#include "llvm/Support/raw_ostream.h"
namespace Carbon::SemIR {
// Manages the chunks created by the formatter.
//
// All output of the formatter is stored as `OutputChunk`s. Tentative scopes,
// such as constants, may have output for instructions made optional; these are
// stored as `include_in_output=false`, but have dependencies in case they're
// later referenced. Unreferenced tentative chunks are omitted from the output.
//
// Output of the main file scope is always included in output, and causes
// referenced tentative instructions to be included in output, including
// indirect dependencies. During this, `Formatter::FormatName` will mark
// referenced instructions for inclusion.
class FormatterChunks {
public:
struct ChunkId {
size_t index;
};
// A chunk of the buffered output.
struct OutputChunk {
// Whether this chunk is known to be included in the output.
bool include_in_output;
// The textual contents of this chunk.
std::string chunk = std::string();
// Indices in `ouput_chunks_` that should be included in the output if this
// one is.
llvm::SmallVector<ChunkId> dependencies = {};
};
// All formatted output within the scope of this object is redirected to a
// new tentative `OutputChunk`. The new chunk will depend on `parent_chunk`.
struct TentativeScope {
explicit TentativeScope(FormatterChunks* chunks, ChunkId parent_chunk);
~TentativeScope();
FormatterChunks* chunks;
ChunkId chunk;
};
// Flushes the buffered output to the current chunk.
auto FlushChunk() -> void;
// Adds a new chunk to the output. Does not flush existing output, so should
// only be called if there is no buffered output.
auto AddChunkNoFlush(bool include_in_output) -> ChunkId;
// Flushes the current chunk and add a new chunk to the output.
auto AddChunk(bool include_in_output) -> ChunkId;
// Marks the given chunk as being included in the output if the current chunk
// is.
auto IncludeChunkInOutput(ChunkId chunk) -> void;
// Write buffered output to the given stream.
auto Write(llvm::raw_ostream& stream) -> void;
auto out() -> llvm::raw_ostream& { return out_; }
private:
friend struct TentativeOutputScope;
// The output stream buffer.
std::string buffer_;
// The output stream.
llvm::raw_string_ostream out_{buffer_};
// Chunks of output text that we have created so far.
llvm::SmallVector<OutputChunk> output_chunks_;
};
} // namespace Carbon::SemIR
#endif // CARBON_TOOLCHAIN_SEM_IR_FORMATTER_CHUNKS_H_